diff --git a/README.md b/README.md index 6fc5b9ef67..6e33b33fab 100644 --- a/README.md +++ b/README.md @@ -167,7 +167,7 @@ if __name__ == '__main__': ### More examples -Explore our [Examples](https://crawlee.dev/python/docs/examples) page in the Crawlee documentation for a wide range of additional use cases and demonstrations. +Explore the [Guides](https://crawlee.dev/python/docs/guides) section of the Crawlee documentation for a wide range of additional use cases and demonstrations. ## Features diff --git a/docs/quick-start/code_examples/beautifulsoup_crawler_example.py b/docs/01_quick-start/code_examples/beautifulsoup_crawler_example.py similarity index 100% rename from docs/quick-start/code_examples/beautifulsoup_crawler_example.py rename to docs/01_quick-start/code_examples/beautifulsoup_crawler_example.py diff --git a/docs/quick-start/code_examples/parsel_crawler_example.py b/docs/01_quick-start/code_examples/parsel_crawler_example.py similarity index 100% rename from docs/quick-start/code_examples/parsel_crawler_example.py rename to docs/01_quick-start/code_examples/parsel_crawler_example.py diff --git a/docs/quick-start/code_examples/playwright_crawler_example.py b/docs/01_quick-start/code_examples/playwright_crawler_example.py similarity index 100% rename from docs/quick-start/code_examples/playwright_crawler_example.py rename to docs/01_quick-start/code_examples/playwright_crawler_example.py diff --git a/docs/quick-start/code_examples/playwright_crawler_headful_example.py b/docs/01_quick-start/code_examples/playwright_crawler_headful_example.py similarity index 100% rename from docs/quick-start/code_examples/playwright_crawler_headful_example.py rename to docs/01_quick-start/code_examples/playwright_crawler_headful_example.py diff --git a/docs/quick-start/index.mdx b/docs/01_quick-start/index.mdx similarity index 92% rename from docs/quick-start/index.mdx rename to docs/01_quick-start/index.mdx index 8125f497d6..c8b61ec0eb 100644 --- a/docs/quick-start/index.mdx +++ b/docs/01_quick-start/index.mdx @@ -1,6 +1,7 @@ --- id: quick-start title: Quick start +description: Build and run your first Crawlee crawler in a few minutes, and pick the crawler type that fits your project. --- import ApiLink from '@site/src/components/ApiLink'; @@ -15,7 +16,7 @@ import PlaywrightCrawlerExample from '!!raw-loader!roa-loader!./code_examples/pl import PlaywrightCrawlerHeadfulExample from '!!raw-loader!./code_examples/playwright_crawler_headful_example.py'; -This short tutorial will help you start scraping with Crawlee in just a minute or two. For an in-depth understanding of how Crawlee works, check out the [Introduction](../introduction/index.mdx) section, which provides a comprehensive step-by-step guide to creating your first scraper. +This short tutorial will help you start scraping with Crawlee in just a minute or two. For an in-depth understanding of how Crawlee works, check out the [Introduction](../02_introduction/index.mdx) section, which provides a comprehensive step-by-step guide to creating your first scraper. ## Choose your crawler @@ -61,7 +62,7 @@ If you plan to use the `PlaywrightCrawler` playwright install ``` -For detailed installation instructions, see the [Setting up](../introduction/01_setting_up.mdx) documentation page. +For detailed installation instructions, see the [Setting up](../02_introduction/01_setting_up.mdx) documentation page. ## Crawling @@ -128,6 +129,6 @@ If you want to change the storage directory, you can set the `CRAWLEE_STORAGE_DI ## Examples and further reading -For more examples showcasing various features of Crawlee, visit the [Examples](/docs/examples) section of the documentation. To get a deeper understanding of Crawlee and its components, read the step-by-step [Introduction](../introduction/index.mdx) guide. +For more examples showcasing various features of Crawlee, visit the [Guides](./guides) section of the documentation. To get a deeper understanding of Crawlee and its components, read the step-by-step [Introduction](../02_introduction/index.mdx) guide. [//]: # (TODO: add related links once they are ready) diff --git a/docs/introduction/01_setting_up.mdx b/docs/02_introduction/01_setting_up.mdx similarity index 98% rename from docs/introduction/01_setting_up.mdx rename to docs/02_introduction/01_setting_up.mdx index 4c5215a576..75858d517c 100644 --- a/docs/introduction/01_setting_up.mdx +++ b/docs/02_introduction/01_setting_up.mdx @@ -1,6 +1,7 @@ --- id: setting-up title: Setting up +description: How to install Crawlee, set up your environment, and bootstrap a new project. --- import ApiLink from '@site/src/components/ApiLink'; diff --git a/docs/introduction/02_first_crawler.mdx b/docs/02_introduction/02_first_crawler.mdx similarity index 98% rename from docs/introduction/02_first_crawler.mdx rename to docs/02_introduction/02_first_crawler.mdx index 0e8d5bd6e0..17e77345e0 100644 --- a/docs/introduction/02_first_crawler.mdx +++ b/docs/02_introduction/02_first_crawler.mdx @@ -1,6 +1,7 @@ --- id: first-crawler title: First crawler +description: Build your first Crawlee crawler - set up a request queue, write a request handler, and crawl your first page. --- import ApiLink from '@site/src/components/ApiLink'; diff --git a/docs/introduction/03_adding_more_urls.mdx b/docs/02_introduction/03_adding_more_urls.mdx similarity index 98% rename from docs/introduction/03_adding_more_urls.mdx rename to docs/02_introduction/03_adding_more_urls.mdx index bfe2337d07..b183be04b5 100644 --- a/docs/introduction/03_adding_more_urls.mdx +++ b/docs/02_introduction/03_adding_more_urls.mdx @@ -1,6 +1,7 @@ --- id: adding-more-urls title: Adding more URLs +description: Grow the crawl by discovering and enqueuing new links, with filtering and deduplication handled for you. --- import ApiLink from '@site/src/components/ApiLink'; diff --git a/docs/introduction/04_real_world_project.mdx b/docs/02_introduction/04_real_world_project.mdx similarity index 98% rename from docs/introduction/04_real_world_project.mdx rename to docs/02_introduction/04_real_world_project.mdx index 61f6435980..21910a488d 100644 --- a/docs/introduction/04_real_world_project.mdx +++ b/docs/02_introduction/04_real_world_project.mdx @@ -1,6 +1,7 @@ --- id: real-world-project title: Real-world project +description: Plan a real scraping project - choose the data to collect and analyze the target website before writing code. --- import ApiLink from '@site/src/components/ApiLink'; diff --git a/docs/introduction/05_crawling.mdx b/docs/02_introduction/05_crawling.mdx similarity index 96% rename from docs/introduction/05_crawling.mdx rename to docs/02_introduction/05_crawling.mdx index 7c68662766..412db44d82 100644 --- a/docs/introduction/05_crawling.mdx +++ b/docs/02_introduction/05_crawling.mdx @@ -1,6 +1,7 @@ --- id: crawling title: Crawling +description: Crawl the example Warehouse store - visit the category listings and enqueue the product detail pages. --- import ApiLink from '@site/src/components/ApiLink'; diff --git a/docs/introduction/06_scraping.mdx b/docs/02_introduction/06_scraping.mdx similarity index 98% rename from docs/introduction/06_scraping.mdx rename to docs/02_introduction/06_scraping.mdx index 51c86e5835..cd95249009 100644 --- a/docs/introduction/06_scraping.mdx +++ b/docs/02_introduction/06_scraping.mdx @@ -1,6 +1,7 @@ --- id: scraping title: Scraping +description: Extract structured data such as titles, prices, and stock information from the product detail pages. --- import ApiLink from '@site/src/components/ApiLink'; diff --git a/docs/introduction/07_saving_data.mdx b/docs/02_introduction/07_saving_data.mdx similarity index 86% rename from docs/introduction/07_saving_data.mdx rename to docs/02_introduction/07_saving_data.mdx index adddd93af9..7374f89645 100644 --- a/docs/introduction/07_saving_data.mdx +++ b/docs/02_introduction/07_saving_data.mdx @@ -1,6 +1,7 @@ --- id: saving-data title: Saving data +description: Persist the scraped results into a dataset and find them on disk. --- import ApiLink from '@site/src/components/ApiLink'; @@ -88,19 +89,10 @@ A helper `context.push_data` save :::info Automatic dataset initialization -Each time you start Crawlee a default `Dataset` is automatically created, so there's no need to initialize it or create an instance first. You can create as many datasets as you want and even give them names. For more details see the `Dataset.open` function. +Each time you start Crawlee a default `Dataset` is automatically created, so there's no need to initialize it or create an instance first. You can create as many datasets as you want and even give them names. For more details see the [Storages](../concepts/storages#dataset) page and the `Dataset.open` function. ::: -{/* TODO: mention result storage guide once it's done - -:::info Automatic dataset initialization - -Each time you start Crawlee a default `Dataset` is automatically created, so there's no need to initialize it or create an instance first. You can create as many datasets as you want and even give them names. For more details see the [Result storage guide](../guides/result-storage#dataset) and the `Dataset.open()` function. - -::: -*/} - ## Finding saved data Unless you changed the configuration that Crawlee uses locally, which would suggest that you knew what you were doing, and you didn't need this tutorial anyway, you'll find your data in the storage directory that Crawlee creates in the working directory of the running script: @@ -111,16 +103,12 @@ Unless you changed the configuration that Crawlee uses locally, which would sugg The above folder will hold all your saved data in numbered files, as they were pushed into the dataset. Each file represents one invocation of `Dataset.push_data` or one table row. -{/* TODO: add mention of "Result storage guide" once it's ready: - :::tip Single file data storage options -If you would like to store your data in a single big file, instead of many small ones, see the [Result storage guide](../guides/result-storage#key-value-store) for Key-value stores. +If you would like to store your data in a single big file, instead of many small ones, see how to [export the whole dataset](../concepts/storages#exporting-the-dataset) to JSON or CSV, or use a [key-value store](../concepts/storages#key-value-store). ::: -*/} - ## Next steps Next, you'll see some improvements that you can add to your crawler code that will make it more readable and maintainable in the long run. diff --git a/docs/introduction/08_refactoring.mdx b/docs/02_introduction/08_refactoring.mdx similarity index 94% rename from docs/introduction/08_refactoring.mdx rename to docs/02_introduction/08_refactoring.mdx index a194a9e839..7245e576f3 100644 --- a/docs/introduction/08_refactoring.mdx +++ b/docs/02_introduction/08_refactoring.mdx @@ -1,6 +1,7 @@ --- id: refactoring title: Refactoring +description: Clean up the crawler code with a router and separate handlers to keep the project maintainable. --- import ApiLink from '@site/src/components/ApiLink'; @@ -21,7 +22,7 @@ You might be wondering about the **anti-blocking, bot-protection avoiding stealt However, the default configuration, while powerful, may not cover every scenario. -If you want to learn more, browse the [Avoid getting blocked](../guides/avoid-blocking), [Proxy management](../guides/proxy-management) and [Session management](../guides/session-management) guides. +If you want to learn more, browse the [Avoid getting blocked](../guides/avoid-blocking), [Proxy management](../concepts/proxy-management) and [Session management](../concepts/session-management) guides. */} To promote good coding practices, let's look at how you can use a `Router` class to better structure your crawler code. diff --git a/docs/introduction/09_running_in_cloud.mdx b/docs/02_introduction/09_running_in_cloud.mdx similarity index 98% rename from docs/introduction/09_running_in_cloud.mdx rename to docs/02_introduction/09_running_in_cloud.mdx index db8273f94f..c6fb8a5859 100644 --- a/docs/introduction/09_running_in_cloud.mdx +++ b/docs/02_introduction/09_running_in_cloud.mdx @@ -2,7 +2,7 @@ id: deployment title: Running your crawler in the Cloud sidebar_label: Running in the Cloud -description: Deploying Crawlee-python projects to the Apify platform +description: Deploy your Crawlee for Python project to the Apify platform and run it in the cloud. --- import CodeBlock from '@theme/CodeBlock'; diff --git a/docs/introduction/code_examples/02_bs.py b/docs/02_introduction/code_examples/02_bs.py similarity index 100% rename from docs/introduction/code_examples/02_bs.py rename to docs/02_introduction/code_examples/02_bs.py diff --git a/docs/introduction/code_examples/02_bs_better.py b/docs/02_introduction/code_examples/02_bs_better.py similarity index 100% rename from docs/introduction/code_examples/02_bs_better.py rename to docs/02_introduction/code_examples/02_bs_better.py diff --git a/docs/introduction/code_examples/02_request_queue.py b/docs/02_introduction/code_examples/02_request_queue.py similarity index 100% rename from docs/introduction/code_examples/02_request_queue.py rename to docs/02_introduction/code_examples/02_request_queue.py diff --git a/docs/introduction/code_examples/03_enqueue_strategy.py b/docs/02_introduction/code_examples/03_enqueue_strategy.py similarity index 100% rename from docs/introduction/code_examples/03_enqueue_strategy.py rename to docs/02_introduction/code_examples/03_enqueue_strategy.py diff --git a/docs/introduction/code_examples/03_finding_new_links.py b/docs/02_introduction/code_examples/03_finding_new_links.py similarity index 100% rename from docs/introduction/code_examples/03_finding_new_links.py rename to docs/02_introduction/code_examples/03_finding_new_links.py diff --git a/docs/introduction/code_examples/03_globs.py b/docs/02_introduction/code_examples/03_globs.py similarity index 100% rename from docs/introduction/code_examples/03_globs.py rename to docs/02_introduction/code_examples/03_globs.py diff --git a/docs/introduction/code_examples/03_original_code.py b/docs/02_introduction/code_examples/03_original_code.py similarity index 100% rename from docs/introduction/code_examples/03_original_code.py rename to docs/02_introduction/code_examples/03_original_code.py diff --git a/docs/introduction/code_examples/03_transform_request.py b/docs/02_introduction/code_examples/03_transform_request.py similarity index 100% rename from docs/introduction/code_examples/03_transform_request.py rename to docs/02_introduction/code_examples/03_transform_request.py diff --git a/docs/introduction/code_examples/04_sanity_check.py b/docs/02_introduction/code_examples/04_sanity_check.py similarity index 100% rename from docs/introduction/code_examples/04_sanity_check.py rename to docs/02_introduction/code_examples/04_sanity_check.py diff --git a/docs/introduction/code_examples/05_crawling_detail.py b/docs/02_introduction/code_examples/05_crawling_detail.py similarity index 100% rename from docs/introduction/code_examples/05_crawling_detail.py rename to docs/02_introduction/code_examples/05_crawling_detail.py diff --git a/docs/introduction/code_examples/05_crawling_listing.py b/docs/02_introduction/code_examples/05_crawling_listing.py similarity index 100% rename from docs/introduction/code_examples/05_crawling_listing.py rename to docs/02_introduction/code_examples/05_crawling_listing.py diff --git a/docs/introduction/code_examples/06_scraping.py b/docs/02_introduction/code_examples/06_scraping.py similarity index 100% rename from docs/introduction/code_examples/06_scraping.py rename to docs/02_introduction/code_examples/06_scraping.py diff --git a/docs/introduction/code_examples/07_final_code.py b/docs/02_introduction/code_examples/07_final_code.py similarity index 100% rename from docs/introduction/code_examples/07_final_code.py rename to docs/02_introduction/code_examples/07_final_code.py diff --git a/docs/introduction/code_examples/07_first_code.py b/docs/02_introduction/code_examples/07_first_code.py similarity index 100% rename from docs/introduction/code_examples/07_first_code.py rename to docs/02_introduction/code_examples/07_first_code.py diff --git a/docs/introduction/code_examples/08_main.py b/docs/02_introduction/code_examples/08_main.py similarity index 100% rename from docs/introduction/code_examples/08_main.py rename to docs/02_introduction/code_examples/08_main.py diff --git a/docs/introduction/code_examples/08_routes.py b/docs/02_introduction/code_examples/08_routes.py similarity index 100% rename from docs/introduction/code_examples/08_routes.py rename to docs/02_introduction/code_examples/08_routes.py diff --git a/docs/introduction/code_examples/09_apify_sdk.py b/docs/02_introduction/code_examples/09_apify_sdk.py similarity index 100% rename from docs/introduction/code_examples/09_apify_sdk.py rename to docs/02_introduction/code_examples/09_apify_sdk.py diff --git a/docs/guides/code_examples/http_crawlers/__init__.py b/docs/02_introduction/code_examples/__init__.py similarity index 100% rename from docs/guides/code_examples/http_crawlers/__init__.py rename to docs/02_introduction/code_examples/__init__.py diff --git a/docs/introduction/code_examples/routes.py b/docs/02_introduction/code_examples/routes.py similarity index 100% rename from docs/introduction/code_examples/routes.py rename to docs/02_introduction/code_examples/routes.py diff --git a/docs/introduction/index.mdx b/docs/02_introduction/index.mdx similarity index 94% rename from docs/introduction/index.mdx rename to docs/02_introduction/index.mdx index 138c6fddb0..fc7d3dab86 100644 --- a/docs/introduction/index.mdx +++ b/docs/02_introduction/index.mdx @@ -1,6 +1,7 @@ --- id: introduction title: Introduction +description: A step-by-step tutorial that takes you from your first crawler to a production-ready scraper for a real website. --- import ApiLink from '@site/src/components/ApiLink'; diff --git a/docs/guides/architecture_overview.mdx b/docs/03_concepts/01_architecture_overview.mdx similarity index 91% rename from docs/guides/architecture_overview.mdx rename to docs/03_concepts/01_architecture_overview.mdx index 672e648d36..21b2b7ef7c 100644 --- a/docs/guides/architecture_overview.mdx +++ b/docs/03_concepts/01_architecture_overview.mdx @@ -6,7 +6,7 @@ description: An overview of the core components of the Crawlee library and its a import ApiLink from '@site/src/components/ApiLink'; -Crawlee is a modern and modular web scraping framework. It is designed for both HTTP-only and browser-based scraping. In this guide, we will provide a high-level overview of its architecture and the main components that make up the system. +Crawlee is a modern and modular web scraping framework. It is designed for both HTTP-only and browser-based scraping. This page provides a high-level overview of its architecture and the main components that make up the system. ## Crawler @@ -76,7 +76,7 @@ PlaywrightCrawler --|> StagehandCrawler ### HTTP crawlers -HTTP crawlers use HTTP clients to fetch pages and parse them with HTML parsing libraries. They are fast and efficient for sites that do not require JavaScript rendering. HTTP clients are Crawlee components that wrap around HTTP libraries like [httpx2](https://httpx2.pydantic.dev/), [curl-impersonate](https://github.com/lwthiker/curl-impersonate) or [impit](https://github.com/apify/impit) and handle HTTP communication for requests and responses. You can learn more about them in the [HTTP clients guide](./http-clients). +HTTP crawlers use HTTP clients to fetch pages and parse them with HTML parsing libraries. They are fast and efficient for sites that do not require JavaScript rendering. HTTP clients are Crawlee components that wrap around HTTP libraries like [httpx2](https://httpx2.pydantic.dev/), [curl-impersonate](https://github.com/lwthiker/curl-impersonate) or [impit](https://github.com/apify/impit) and handle HTTP communication for requests and responses. You can learn more about them on the [HTTP clients](./http-clients) page. HTTP crawlers inherit from `AbstractHttpCrawler` and there are five crawlers that belong to this category: @@ -86,18 +86,18 @@ HTTP crawlers inherit from `AbstractHttp - `PydanticAiCrawler` parses HTML with Parsel and uses an LLM to extract structured data into a validated Pydantic model. - `FileDownloadCrawler` downloads files of any content type, optionally streaming large bodies in chunks. -You can learn more about HTTP crawlers in the [HTTP crawlers guide](./http-crawlers). +You can learn more about HTTP crawlers on the [HTTP crawlers](./http-crawlers) page. ### Browser crawlers Browser crawlers use a real browser to render pages, enabling scraping of sites that require JavaScript. They manage browser instances, pages, and context lifecycles. Crawlee provides two browser crawlers: -- `PlaywrightCrawler` utilizes the [Playwright](https://playwright.dev/) library and provides a high-level API for controlling and navigating browsers. You can learn more about it in the [Playwright crawler guide](./playwright-crawler). -- `StagehandCrawler` extends `PlaywrightCrawler` with AI-powered browser automation via [Stagehand](https://github.com/browserbase/stagehand). It adds natural-language methods (`act`, `extract`, `observe`, `execute`) directly on the page object. You can learn more about it in the [Stagehand crawler guide](./stagehand-crawler). +- `PlaywrightCrawler` utilizes the [Playwright](https://playwright.dev/) library and provides a high-level API for controlling and navigating browsers. You can learn more about it on the [Playwright crawler](./playwright-crawler) page. +- `StagehandCrawler` extends `PlaywrightCrawler` with AI-powered browser automation via [Stagehand](https://github.com/browserbase/stagehand). It adds natural-language methods (`act`, `extract`, `observe`, `execute`) directly on the page object. You can learn more about it in the [Stagehand crawler guide](../guides/stagehand-crawler). ### Adaptive crawler -The `AdaptivePlaywrightCrawler` sits between HTTP and browser crawlers. It can automatically decide whether to use HTTP or browser crawling for each request based on heuristics or user configuration. This allows for optimal performance and compatibility. It also provides a uniform interface for both crawling types (modes). You can learn more about adaptive crawling in the [Adaptive Playwright crawler guide](./adaptive-playwright-crawler). +The `AdaptivePlaywrightCrawler` sits between HTTP and browser crawlers. It can automatically decide whether to use HTTP or browser crawling for each request based on heuristics or user configuration. This allows for optimal performance and compatibility. It also provides a uniform interface for both crawling types (modes). You can learn more about adaptive crawling on the [Adaptive Playwright crawler](./adaptive-playwright-crawler) page. ## Crawling contexts @@ -208,7 +208,7 @@ Crawlee provides three built-in storage types for managing data: - `KeyValueStore` - Storage for arbitrary data like JSON documents, images or configs. It supports get and set operations with key-value pairs; updates are only possible by replacement. - `RequestQueue` - A managed queue for pending and completed requests, with automatic deduplication and dynamic addition of new items. It is used to track URLs for crawling. -See the [Storages guide](./storages) for more details. +See the [Storages](./storages) page for more details. ```mermaid --- @@ -294,7 +294,7 @@ StorageClient --|> ApifyStorageClient Storage clients can be registered globally with the `ServiceLocator` (you will learn more about the `ServiceLocator` in the next section), passed directly to crawlers, or specified when opening individual storage instances. You can also create custom storage clients by implementing the `StorageClient` interface. -See the [Storage clients guide](./storage-clients) for more details. +See the [Storage clients](./storage-clients) page for more details. ## Request router @@ -312,7 +312,7 @@ The request routing in Crawlee supports: - Failed request handlers - Handle requests that exceed retry limits. - Pre-navigation hooks - Execute logic before navigating to URLs. -See the [Request router guide](./request-router) for detailed information and examples. +See the [Request router](./request-router) page for detailed information and examples. ## Service locator @@ -324,7 +324,7 @@ The `ServiceLocator` is a central r Services can be registered globally through the `service_locator` singleton instance, passed to crawler constructors, or provided when opening individual storage instances. The service locator includes conflict prevention mechanisms to ensure configuration consistency and prevent accidental service conflicts during runtime. -See the [Service locator guide](./service-locator) for detailed information about service registration and configuration options. +See the [Service locator](./service-locator) page for detailed information about service registration and configuration options. ## Request loaders @@ -342,7 +342,7 @@ Request loaders provide a subset of `RequestQue Request loaders are useful when you need to start with a predefined set of URLs. The tandem approach allows processing requests from static sources (like files or sitemaps) while maintaining the ability to add new requests dynamically. -See the [Request loaders guide](./request-loaders) for detailed information. +See the [Request loaders](./request-loaders) page for detailed information. ## Event manager @@ -356,7 +356,7 @@ Crawlee provides several implementations of the event manager: :::info -You can learn more about `Snapshotter` and `AutoscaledPool` and their configuration in the [Scaling crawlers guide](./scaling-crawlers). +You can learn more about `Snapshotter` and `AutoscaledPool` and their configuration on the [Scaling crawlers](./scaling-crawlers) page. ::: @@ -412,7 +412,7 @@ The core component of session management in Crawlee is Statistics are logged at configurable intervals in both table and inline formats, with final summary data returned from the `crawler.run` method available through `FinalStatistics`. -## Conclusion - -In this guide, we provided a high-level overview of the core components of the Crawlee library and its architecture. We covered the main components like crawlers, crawling contexts, storages, request routers, service locator, request loaders, event manager, session management, and statistics. Check out other guides, the [API reference](https://crawlee.dev/python/api), and [Examples](../examples) for more details on how to use these components in your own projects. - -If you have questions or need assistance, feel free to reach out on our [GitHub](https://github.com/apify/crawlee-python) or join our [Discord community](https://discord.com/invite/jyEM2PRvMU). Happy scraping! diff --git a/docs/guides/http_crawlers.mdx b/docs/03_concepts/02_http_crawlers.mdx similarity index 85% rename from docs/guides/http_crawlers.mdx rename to docs/03_concepts/02_http_crawlers.mdx index 00100d5e46..e2e35437f7 100644 --- a/docs/guides/http_crawlers.mdx +++ b/docs/03_concepts/02_http_crawlers.mdx @@ -20,17 +20,20 @@ import LexborParser from '!!raw-loader!roa-loader!./code_examples/http_crawlers/ import PyqueryParser from '!!raw-loader!roa-loader!./code_examples/http_crawlers/pyquery_parser.py'; import ScraplingParser from '!!raw-loader!roa-loader!./code_examples/http_crawlers/scrapling_parser.py'; +import FileDownloadExample from '!!raw-loader!roa-loader!./code_examples/http_crawlers/file_download.py'; +import FileDownloadStreamExample from '!!raw-loader!./code_examples/http_crawlers/file_download_stream.py'; + import SelectolaxParserSource from '!!raw-loader!./code_examples/http_crawlers/selectolax_parser.py'; import SelectolaxContextSource from '!!raw-loader!./code_examples/http_crawlers/selectolax_context.py'; import SelectolaxCrawlerSource from '!!raw-loader!./code_examples/http_crawlers/selectolax_crawler.py'; import SelectolaxCrawlerRunSource from '!!raw-loader!./code_examples/http_crawlers/selectolax_crawler_run.py'; import AdaptiveCrawlerRunSource from '!!raw-loader!./code_examples/http_crawlers/selectolax_adaptive_run.py'; -HTTP crawlers are ideal for extracting data from server-rendered websites that don't require JavaScript execution. These crawlers make requests via HTTP clients to fetch HTML content and then parse it using various parsing libraries. For client-side rendered content, where you need to execute JavaScript consider using [Playwright crawler](https://crawlee.dev/python/docs/guides/playwright-crawler) instead. +HTTP crawlers are ideal for extracting data from server-rendered websites that don't require JavaScript execution. These crawlers make requests via HTTP clients to fetch HTML content and then parse it using various parsing libraries. For client-side rendered content, where you need to execute JavaScript, consider using the [Playwright crawler](./playwright-crawler) instead. ## Overview -All HTTP crawlers share a common architecture built around the `AbstractHttpCrawler` base class. The main differences lie in the parsing strategy and the context provided to request handlers. There are `BeautifulSoupCrawler`, `ParselCrawler`, `HttpCrawler`, and `FileDownloadCrawler`. It can also be extended to create custom crawlers with specialized parsing requirements. They use HTTP clients to fetch page content and parsing libraries to extract data from the HTML, check out the [HTTP clients guide](./http-clients) to learn about the HTTP clients used by these crawlers, how to switch between them, and how to create custom HTTP clients tailored to your specific requirements. +All HTTP crawlers share a common architecture built around the `AbstractHttpCrawler` base class. The main differences lie in the parsing strategy and the context provided to request handlers. There are `BeautifulSoupCrawler`, `ParselCrawler`, `HttpCrawler`, and `FileDownloadCrawler`. It can also be extended to create custom crawlers with specialized parsing requirements. They use HTTP clients to fetch page content and parsing libraries to extract data from the HTML, check out the [HTTP clients](./http-clients) page to learn about the HTTP clients used by these crawlers, how to switch between them, and how to create custom HTTP clients tailored to your specific requirements. ```mermaid --- @@ -136,7 +139,19 @@ The following examples demonstrate how to integrate with several popular parsing ## FileDownloadCrawler -The `FileDownloadCrawler` downloads files instead of scraping pages. It accepts any content type without parsing and gives the request handler direct access to the response body. By default the whole file is buffered in memory. For large files, construct the crawler with `stream=True` and consume the body in chunks via `read_stream()`. For usage, see the [Download files](../examples/file-download) example. +The `FileDownloadCrawler` downloads files such as PDFs, images or videos instead of scraping pages. It accepts any content type without parsing and gives the request handler direct access to the response body. Each downloaded file can be saved to the `KeyValueStore` together with the content type reported by the server. + + + {FileDownloadExample} + + +### Streaming large files + +By default the whole file is buffered in memory, which doesn't scale to large downloads. Construct the crawler with `stream=True` and the request handler receives a response whose body hasn't been read yet. Consume it in chunks with `read_stream()` and write each chunk to disk as it arrives. + + + {FileDownloadStreamExample} + ## Custom HTTP crawler @@ -192,9 +207,3 @@ The custom crawler works like any built-in crawler. Request handlers receive you - -## Conclusion - -This guide provided a comprehensive overview of HTTP crawlers in Crawlee. You learned about the three main crawler types - `BeautifulSoupCrawler` for fault-tolerant HTML parsing, `ParselCrawler` for high-performance extraction with XPath and CSS selectors, and `HttpCrawler` for raw response processing. You also discovered how to integrate third-party parsing libraries with `HttpCrawler` and how to create fully custom crawlers using `AbstractHttpCrawler` for specialized parsing requirements. - -If you have questions or need assistance, feel free to reach out on our [GitHub](https://github.com/apify/crawlee-python) or join our [Discord community](https://discord.com/invite/jyEM2PRvMU). Happy scraping! diff --git a/docs/guides/playwright_crawler.mdx b/docs/03_concepts/03_playwright_crawler.mdx similarity index 75% rename from docs/guides/playwright_crawler.mdx rename to docs/03_concepts/03_playwright_crawler.mdx index 26a03c79e2..c291de3be4 100644 --- a/docs/guides/playwright_crawler.mdx +++ b/docs/03_concepts/03_playwright_crawler.mdx @@ -8,12 +8,15 @@ import ApiLink from '@site/src/components/ApiLink'; import CodeBlock from '@theme/CodeBlock'; import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; +import BasicExample from '!!raw-loader!roa-loader!./code_examples/playwright_crawler/playwright_crawler.py'; import MultipleLaunchExample from '!!raw-loader!roa-loader!./code_examples/playwright_crawler/multiple_launch_example.py'; import BrowserConfigurationExample from '!!raw-loader!roa-loader!./code_examples/playwright_crawler/browser_configuration_example.py'; import NavigationHooksExample from '!!raw-loader!roa-loader!./code_examples/playwright_crawler/navigation_hooks_example.py'; import BrowserPoolLaunchHooksExample from '!!raw-loader!roa-loader!./code_examples/playwright_crawler/browser_pool_launch_hooks_example.py'; import BrowserPoolPageHooksExample from '!!raw-loader!roa-loader!./code_examples/playwright_crawler/browser_pool_page_hooks_example.py'; import PluginBrowserConfigExample from '!!raw-loader!./code_examples/playwright_crawler/plugin_browser_configuration_example.py'; +import BlockRequestsExample from '!!raw-loader!roa-loader!./code_examples/playwright_crawler/playwright_block_requests.py'; +import CaptureScreenshotExample from '!!raw-loader!roa-loader!./code_examples/playwright_crawler/capture_screenshot_using_playwright.py'; A `PlaywrightCrawler` is a browser-based crawler. In contrast to HTTP-based crawlers like `ParselCrawler` or `BeautifulSoupCrawler`, it uses a real browser to render pages and extract data. It is built on top of the [Playwright](https://playwright.dev/python/) browser automation library. While browser-based crawlers are typically slower and less efficient than HTTP-based crawlers, they can handle dynamic, client-side rendered sites that standard HTTP-based crawlers cannot manage. @@ -25,7 +28,15 @@ Use `PlaywrightCrawler` in scena - **Anti-scraping protection**: Helpful for sites using JavaScript-based security or advanced anti-automation measures. - **Complex cookie management**: Necessary for sites with session or cookie requirements that standard HTTP-based crawlers cannot handle easily. -If [HTTP-based crawlers](https://crawlee.dev/python/docs/guides/http-crawlers) are insufficient, `PlaywrightCrawler` can address these challenges. See a [basic example](../examples/playwright-crawler) for a typical usage demonstration. +If [HTTP-based crawlers](./http-crawlers) are insufficient, `PlaywrightCrawler` can address these challenges. + +## Basic example + +The following example crawls the Hacker News website using headless Chromium. The crawler manages the browser and page instances for you. In the request handler, Playwright's API is used to extract the title, rank, and URL of each post, and links to the next pages are enqueued to keep the crawl going. + + + {BasicExample} + ## Advanced configuration @@ -64,7 +75,7 @@ Create a subclass only when an integration uses a different browser launch API. Keep the inherited plugin configuration intact so that browser context options remain available to `BrowserPool`. When constructing the controller, explicitly preserve or intentionally replace `use_incognito_pages`, `max_open_pages_per_browser`, and fingerprint or header generation behavior. The custom launch path must likewise define how it applies browser launch options and `user_data_dir`. -See the [Camoufox example](../examples/playwright-crawler-with-camoufox) for a complete implementation that uses a different browser launcher. +See the [Camoufox example](../guides/avoid-blocking#using-camoufox) for a complete implementation that uses a different browser launcher. ## Browser pool lifecycle hooks @@ -88,12 +99,30 @@ For additional setup or event-driven actions around page creation and closure, t ## Navigation hooks -Navigation hooks allow for additional configuration at specific points during page navigation. The `pre_navigation_hook` is called before each navigation and provides `PlaywrightPreNavCrawlingContext` - including the [page](https://playwright.dev/python/docs/api/class-page) instance and a `block_requests` helper for filtering unwanted resource types and URL patterns. See the [block requests example](https://crawlee.dev/python/docs/examples/playwright-crawler-with-block-requests) for a dedicated walkthrough. Similarly, the `post_navigation_hook` is called after each navigation and provides `PlaywrightPostNavCrawlingContext` - useful for post-load checks such as detecting CAPTCHAs or verifying page state. +Navigation hooks allow for additional configuration at specific points during page navigation. The `pre_navigation_hook` is called before each navigation and provides `PlaywrightPreNavCrawlingContext` - including the [page](https://playwright.dev/python/docs/api/class-page) instance and a `block_requests` helper for filtering unwanted resource types and URL patterns, covered in [Blocking network requests](#blocking-network-requests). Similarly, the `post_navigation_hook` is called after each navigation and provides `PlaywrightPostNavCrawlingContext` - useful for post-load checks such as detecting CAPTCHAs or verifying page state. {NavigationHooksExample} -## Conclusion +## Blocking network requests + +Loading non-essential resources like images, styles, or analytics scripts wastes bandwidth and slows down crawling. The `block_requests` helper provides the most efficient way to block such requests, as it operates directly in the browser. By default, it blocks URLs matching the following patterns: + +```python +['.css', '.webp', '.jpg', '.jpeg', '.png', '.svg', '.gif', '.woff', '.pdf', '.zip'] +``` + +You can replace the default patterns with your own by providing `url_patterns`, or extend them by passing additional patterns in `extra_url_patterns`. -This guide introduced the `PlaywrightCrawler` and explained how to configure it using `BrowserPool` and `PlaywrightBrowserPlugin`. You learned how to launch multiple browsers, configure browser and context settings, use `BrowserPool` lifecycle hooks, and apply navigation hooks. If you have questions or need assistance, feel free to reach out on our [GitHub](https://github.com/apify/crawlee-python) or join our [Discord community](https://discord.com/invite/jyEM2PRvMU). Happy scraping! + + {BlockRequestsExample} + + +## Capturing screenshots + +Playwright's `page.screenshot()` method makes it easy to capture screenshots of the pages your crawler visits. The following example stores each screenshot in the key-value store as a PNG image, under a unique key generated from the page URL. + + + {CaptureScreenshotExample} + diff --git a/docs/guides/playwright_crawler_adaptive.mdx b/docs/03_concepts/04_playwright_crawler_adaptive.mdx similarity index 95% rename from docs/guides/playwright_crawler_adaptive.mdx rename to docs/03_concepts/04_playwright_crawler_adaptive.mdx index 7957b98015..da2cca2caf 100644 --- a/docs/guides/playwright_crawler_adaptive.mdx +++ b/docs/03_concepts/04_playwright_crawler_adaptive.mdx @@ -10,6 +10,7 @@ import Tabs from '@theme/Tabs'; import TabItem from '@theme/TabItem'; import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; +import AdaptivePlaywrightCrawlerBasicExample from '!!raw-loader!roa-loader!./code_examples/playwright_crawler_adaptive/adaptive_playwright_crawler.py'; import AdaptivePlaywrightCrawlerHandler from '!!raw-loader!roa-loader!./code_examples/playwright_crawler_adaptive/handler.py'; import AdaptivePlaywrightCrawlerPreNavHooks from '!!raw-loader!roa-loader!./code_examples/playwright_crawler_adaptive/pre_nav_hooks.py'; @@ -28,6 +29,14 @@ Use `AdaptivePlaywrightCrawler` + {AdaptivePlaywrightCrawlerBasicExample} + + ## Request handler and adaptive context helpers Request handler for `AdaptivePlaywrightCrawler` works on special context type - `AdaptivePlaywrightCrawlingContext`. This context is sometimes created by HTTP-based sub crawler and sometimes by playwright based sub crawler. Due to its dynamic nature, you can't always access [page](https://playwright.dev/python/docs/api/class-page) object. To overcome this limitation, there are three helper methods on this context that can be called regardless of how the context was created. diff --git a/docs/guides/request_router.mdx b/docs/03_concepts/05_request_router.mdx similarity index 89% rename from docs/guides/request_router.mdx rename to docs/03_concepts/05_request_router.mdx index 1aceab2637..204a04da25 100644 --- a/docs/guides/request_router.mdx +++ b/docs/03_concepts/05_request_router.mdx @@ -26,7 +26,7 @@ Request handlers are user-defined functions that process individual requests and :::note -The code examples in this guide use `ParselCrawler` for demonstration, but the `Router` works with all crawler types. +The code examples on this page use `ParselCrawler` for demonstration, but the `Router` works with all crawler types. ::: @@ -114,8 +114,3 @@ The `AdaptivePlaywrightCrawler` -## Conclusion - -This guide introduced you to the `Router` class and how to organize your crawling logic. You learned how to use built-in and custom routers, implement request handlers with label-based routing, add middleware with `router.use()`, handle errors with error and failed request handlers, and configure pre-navigation hooks for different crawler types. - -If you have questions or need assistance, feel free to reach out on our [GitHub](https://github.com/apify/crawlee-python) or join our [Discord community](https://discord.com/invite/jyEM2PRvMU). Happy scraping! diff --git a/docs/guides/request_loaders.mdx b/docs/03_concepts/06_request_loaders.mdx similarity index 92% rename from docs/guides/request_loaders.mdx rename to docs/03_concepts/06_request_loaders.mdx index a63393142d..8e08bd9035 100644 --- a/docs/guides/request_loaders.mdx +++ b/docs/03_concepts/06_request_loaders.mdx @@ -17,8 +17,9 @@ import SitemapTandemExample from '!!raw-loader!roa-loader!./code_examples/reques import SitemapExplicitTandemExample from '!!raw-loader!roa-loader!./code_examples/request_loaders/sitemap_tandem_example_explicit.py'; import RlBasicPersistExample from '!!raw-loader!roa-loader!./code_examples/request_loaders/rl_basic_example_with_persist.py'; import SitemapPersistExample from '!!raw-loader!roa-loader!./code_examples/request_loaders/sitemap_example_with_persist.py'; +import SitemapTransformExample from '!!raw-loader!roa-loader!./code_examples/request_loaders/using_sitemap_request_loader.py'; -The [`request_loaders`](https://github.com/apify/crawlee-python/tree/master/src/crawlee/request_loaders) sub-package extends the functionality of the `RequestQueue`, providing additional tools for managing URLs and requests. If you are new to Crawlee and unfamiliar with the `RequestQueue`, consider starting with the [Storages](https://crawlee.dev/python/docs/guides/storages) guide first. Request loaders define how requests are fetched and stored, enabling various use cases such as reading URLs from files, external APIs, or combining multiple sources together. +The [`request_loaders`](https://github.com/apify/crawlee-python/tree/master/src/crawlee/request_loaders) sub-package extends the functionality of the `RequestQueue`, providing additional tools for managing URLs and requests. If you are new to Crawlee and unfamiliar with the `RequestQueue`, consider starting with the [Storages](./storages) page first. Request loaders define how requests are fetched and stored, enabling various use cases such as reading URLs from files, external APIs, or combining multiple sources together. ## Overview @@ -122,7 +123,7 @@ Here is a basic example of working with the `Req The `RequestList` supports state persistence, allowing it to resume from where it left off after interruption. This is particularly useful for long-running crawls or when you need to pause and resume crawling later. -To enable persistence, provide `persist_state_key` and optionally `persist_requests_key` parameters, and disable automatic cleanup by setting `purge_on_start = False` in the configuration. The `persist_state_key` saves the loader's progress, while `persist_requests_key` ensures that the request data doesn't change between runs. For more details on resuming interrupted crawls, see the [Resuming a paused crawl](../examples/resuming-paused-crawl) example. +To enable persistence, provide `persist_state_key` and optionally `persist_requests_key` parameters, and disable automatic cleanup by setting `purge_on_start = False` in the configuration. The `persist_state_key` saves the loader's progress, while `persist_requests_key` ensures that the request data doesn't change between runs. For more details on resuming interrupted crawls, see the [Stopping and resuming crawlers](../guides/stopping-and-resuming-crawlers) guide. {RlBasicPersistExample} @@ -152,6 +153,14 @@ Similarly, the `SitemapRequestLoader` + {SitemapTransformExample} + + ## Request managers The `RequestManager` extends `RequestLoader` with write capabilities. In addition to reading requests, a request manager can add and reclaim them. This is essential for dynamic crawling projects where new URLs may emerge during the crawl process, or when certain requests fail and need to be retried. For more details, refer to the `RequestManager` API reference. @@ -196,8 +205,3 @@ Similar to the `RequestList` example a -## Conclusion - -This guide explained the `request_loaders` sub-package, which extends the functionality of the `RequestQueue` with additional tools for managing URLs and requests. You learned about the `RequestLoader`, `RequestManager`, and `RequestManagerTandem` classes, as well as the `RequestList` and `SitemapRequestLoader` implementations. You also saw practical examples of how to work with these classes to handle various crawling scenarios. - -If you have questions or need assistance, feel free to reach out on our [GitHub](https://github.com/apify/crawlee-python) or join our [Discord community](https://discord.com/invite/jyEM2PRvMU). Happy scraping! diff --git a/docs/guides/storages.mdx b/docs/03_concepts/07_storages.mdx similarity index 84% rename from docs/guides/storages.mdx rename to docs/03_concepts/07_storages.mdx index a4348e28ab..9157d3cb4f 100644 --- a/docs/guides/storages.mdx +++ b/docs/03_concepts/07_storages.mdx @@ -21,6 +21,9 @@ import DatasetBasicExample from '!!raw-loader!roa-loader!./code_examples/storage import DatasetWithCrawlerExample from '!!raw-loader!roa-loader!./code_examples/storages/dataset_with_crawler_example.py'; import DatasetWithCrawlerExplicitExample from '!!raw-loader!roa-loader!./code_examples/storages/dataset_with_crawler_explicit_example.py'; +import ExportJsonExample from '!!raw-loader!roa-loader!./code_examples/storages/export_entire_dataset_to_file_json.py'; +import ExportCsvExample from '!!raw-loader!roa-loader!./code_examples/storages/export_entire_dataset_to_file_csv.py'; + import KvsBasicExample from '!!raw-loader!roa-loader!./code_examples/storages/kvs_basic_example.py'; import KvsWithCrawlerExample from '!!raw-loader!roa-loader!./code_examples/storages/kvs_with_crawler_example.py'; import KvsWithCrawlerExplicitExample from '!!raw-loader!roa-loader!./code_examples/storages/kvs_with_crawler_explicit_example.py'; @@ -28,7 +31,7 @@ import KvsWithCrawlerExplicitExample from '!!raw-loader!roa-loader!./code_exampl import CleaningDoNotPurgeExample from '!!raw-loader!roa-loader!./code_examples/storages/cleaning_do_not_purge_example.py'; import CleaningPurgeExplicitlyExample from '!!raw-loader!roa-loader!./code_examples/storages/cleaning_purge_explicitly_example.py'; -Crawlee offers several storage types for managing and persisting your crawling data. Request-oriented storages, such as the `RequestQueue`, help you store and deduplicate URLs, while result-oriented storages, like `Dataset` and `KeyValueStore`, focus on storing and retrieving scraping results. This guide explains when to use each type, how to interact with them, and how to control their lifecycle. +Crawlee offers several storage types for managing and persisting your crawling data. Request-oriented storages, such as the `RequestQueue`, help you store and deduplicate URLs, while result-oriented storages, like `Dataset` and `KeyValueStore`, focus on storing and retrieving scraping results. This page explains when to use each type, how to interact with them, and how to control their lifecycle. ## Overview @@ -37,7 +40,7 @@ Crawlee's storage system consists of two main layers: - **Storages** (`Dataset`, `KeyValueStore`, `RequestQueue`): High-level interfaces for interacting with different storage types. - **Storage clients** (`MemoryStorageClient`, `FileSystemStorageClient`, etc.): Backend implementations that handle the actual data persistence and management. -For more information about storage clients and their configuration, see the [Storage clients guide](./storage-clients). +For more information about storage clients and their configuration, see the [Storage clients](./storage-clients) page. ```mermaid --- @@ -119,7 +122,7 @@ The following code demonstrates the usage of the `RequestQueue`: - The `add_requests` function allows you to manually add specific URLs to the configured request storage. In this case, you must explicitly provide the URLs you want to be added to the request storage. If you need to specify further details of the request, such as a `label` or `user_data`, you have to pass instances of the `Request` class to the helper. -- The `enqueue_links` function is designed to discover new URLs in the current page and add them to the request storage. It can be used with default settings, requiring no arguments, or you can customize its behavior by specifying link element selectors, choosing different enqueue strategies, or applying include/exclude filters to control which URLs are added. See [Crawl website with relative links](../examples/crawl-website-with-relative-links) example for more details. +- The `enqueue_links` function is designed to discover new URLs in the current page and add them to the request storage. It can be used with default settings, requiring no arguments, or you can customize its behavior by specifying link element selectors, choosing different enqueue strategies, or applying include/exclude filters to control which URLs are added. See the [Enqueue strategies](../guides/crawling-links#enqueue-strategies) section of the Crawling links guide for more details. @@ -140,7 +143,7 @@ The `RequestQueue` implements the `RequestManager` class and implementing its required methods. -For a detailed explanation of the `RequestManager` and other related components, refer to the [Request loaders guide](https://crawlee.dev/python/docs/guides/request-loaders). +For a detailed explanation of the `RequestManager` and other related components, refer to the [Request loaders](./request-loaders) page. ## Dataset @@ -172,6 +175,31 @@ Crawlee provides the following helper function to simplify interactions with the - The `push_data` function allows you to manually add data to the dataset. You can optionally specify the dataset ID or its name. +### Exporting the dataset + +The `BasicCrawler.export_data` method exports the entire default dataset to a single file, in either CSV or JSON format. It also accepts additional keyword arguments so you can fine-tune the underlying `json.dump` or `csv.DictWriter` behavior. The method is available on all crawlers. + + + + + {ExportJsonExample} + + + + + {ExportCsvExample} + + + + +Dataset items don't have to share a schema. Different handlers can push different fields, so the items of one dataset often have different keys. By default, the CSV columns are the keys of the first non-empty item. When a later item has a key the first one doesn't, its value isn't written, and Crawlee logs a warning naming the dropped keys. To use the keys of all items as columns instead, pass `collect_all_keys=True`: + +```python +await crawler.export_data(path='results.csv', collect_all_keys=True) +``` + +No value is dropped in that mode. Collecting all keys means reading the whole dataset before the first row can be written, so only the default mode writes rows as it goes. In both modes, cells for columns an item doesn't have stay empty, or hold the `restval` value if you pass one. + ## Key-value store The `KeyValueStore` is designed to save and retrieve data records or files efficiently. Each record is uniquely identified by a key and is associated with a specific MIME type, making the `KeyValueStore` ideal for tasks like saving web page screenshots, PDFs, or tracking the state of crawlers. @@ -196,7 +224,7 @@ The following code demonstrates the usage of the `RequestQueue` and store and retrieve scraping results using the `Dataset` and `KeyValueStore`. You also learned how to use helper functions to simplify interactions with these storages and how to control storage cleanup behavior. +Note that purging behavior may vary between storage client implementations. For more details on storage configuration and client implementations, see the [Storage clients](./storage-clients) page. -If you have questions or need assistance, feel free to reach out on our [GitHub](https://github.com/apify/crawlee-python) or join our [Discord community](https://discord.com/invite/jyEM2PRvMU). Happy scraping! diff --git a/docs/guides/storage_clients.mdx b/docs/03_concepts/08_storage_clients.mdx similarity index 96% rename from docs/guides/storage_clients.mdx rename to docs/03_concepts/08_storage_clients.mdx index 3186f906c9..ee722b987f 100644 --- a/docs/guides/storage_clients.mdx +++ b/docs/03_concepts/08_storage_clients.mdx @@ -551,8 +551,3 @@ Storage clients can be registered in multiple ways: You can also register different storage clients for each storage instance, allowing you to use different backends for different storages. This is useful when you want to use a fast in-memory storage for `RequestQueue` while persisting scraping results in `Dataset` or `KeyValueStore`. -## Conclusion - -Storage clients in Crawlee provide different backends for data storage. Use `MemoryStorageClient` for testing and fast operations without persistence, or `FileSystemStorageClient` for environments where data needs to persist. You can also create custom storage clients for specialized backends by implementing the `StorageClient` interface. - -If you have questions or need assistance, feel free to reach out on our [GitHub](https://github.com/apify/crawlee-python) or join our [Discord community](https://discord.com/invite/jyEM2PRvMU). Happy scraping! diff --git a/docs/guides/http_clients.mdx b/docs/03_concepts/09_http_clients.mdx similarity index 89% rename from docs/guides/http_clients.mdx rename to docs/03_concepts/09_http_clients.mdx index f79fe1941d..04baecef30 100644 --- a/docs/guides/http_clients.mdx +++ b/docs/03_concepts/09_http_clients.mdx @@ -53,7 +53,7 @@ HttpClient --|> CurlImpersonateHttpClient ## Switching between HTTP clients -Crawlee currently provides three main HTTP clients: `ImpitHttpClient`, which uses the `impit` library, `HttpxHttpClient`, which uses the `httpx2` library with `browserforge` for custom HTTP headers and fingerprints, and `CurlImpersonateHttpClient`, which uses the `curl-cffi` library. You can switch between them by setting the `http_client` parameter when initializing a crawler class. The default HTTP client is `ImpitHttpClient`. For more details on anti-blocking features, see our [avoid getting blocked guide](./avoid-blocking). +Crawlee currently provides three main HTTP clients: `ImpitHttpClient`, which uses the `impit` library, `HttpxHttpClient`, which uses the `httpx2` library with `browserforge` for custom HTTP headers and fingerprints, and `CurlImpersonateHttpClient`, which uses the `curl-cffi` library. You can switch between them by setting the `http_client` parameter when initializing a crawler class. The default HTTP client is `ImpitHttpClient`. For more details on anti-blocking features, see our [avoid getting blocked guide](../guides/avoid-blocking). Below are examples of how to configure the HTTP client for the `ParselCrawler`: @@ -111,8 +111,3 @@ HTTP clients are responsible for several key operations: To create a custom HTTP client, you need to inherit from the `HttpClient` base class and implement all required abstract methods. Your implementation must be async-compatible and include proper cleanup and resource management to work seamlessly with Crawlee's concurrent processing model. -## Conclusion - -This guide introduced you to the HTTP clients available in Crawlee and demonstrated how to switch between them, including their installation requirements and usage examples. You also learned about the responsibilities of HTTP clients and how to implement your own custom HTTP client by inheriting from the `HttpClient` base class. - -If you have questions or need assistance, feel free to reach out on our [GitHub](https://github.com/apify/crawlee-python) or join our [Discord community](https://discord.com/invite/jyEM2PRvMU). Happy scraping! diff --git a/docs/guides/session_management.mdx b/docs/03_concepts/10_session_management.mdx similarity index 100% rename from docs/guides/session_management.mdx rename to docs/03_concepts/10_session_management.mdx diff --git a/docs/guides/cookie_management.mdx b/docs/03_concepts/11_cookie_management.mdx similarity index 96% rename from docs/guides/cookie_management.mdx rename to docs/03_concepts/11_cookie_management.mdx index c730e7ac0d..cfd93421f0 100644 --- a/docs/guides/cookie_management.mdx +++ b/docs/03_concepts/11_cookie_management.mdx @@ -21,7 +21,7 @@ In Crawlee, cookies are stored on the `Session``PlaywrightCrawler`, and keep them across retries and runs. +This page covers how to read and set cookies, seed them on new sessions, handle them with `PlaywrightCrawler`, and keep them across retries and runs. ## Reading and setting cookies @@ -103,4 +103,4 @@ To see restoration in action, define the pool outside the crawler and read a ses {PersistCookies} -With persistence enabled, a login you performed on a previous run can carry over, so you skip re-authenticating on every start. For a full walkthrough of logging in, see the [Logging in with a crawler](./logging-in-with-a-crawler) guide. +With persistence enabled, a login you performed on a previous run can carry over, so you skip re-authenticating on every start. For a full walkthrough of logging in, see the [Logging in with a crawler](../guides/logging-in-with-a-crawler) guide. diff --git a/docs/guides/http_headers.mdx b/docs/03_concepts/12_http_headers.mdx similarity index 91% rename from docs/guides/http_headers.mdx rename to docs/03_concepts/12_http_headers.mdx index c7c148fe24..b7a23ea8a1 100644 --- a/docs/guides/http_headers.mdx +++ b/docs/03_concepts/12_http_headers.mdx @@ -10,7 +10,7 @@ import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; import SetHeadersExample from '!!raw-loader!roa-loader!./code_examples/http_headers/set_headers.py'; import BrowserPageHeadersExample from '!!raw-loader!roa-loader!./code_examples/http_headers/browser_page_headers.py'; -Every request a crawler sends includes [HTTP headers](https://developer.mozilla.org/en-US/docs/Web/HTTP/Reference/Headers). These headers tell the server who is making the request, what content is acceptable, and in what language. The server reads them and decides what to return. The same URL can return different content, a different status code, or a blocked page depending on the headers it sees. This guide covers the headers that shape a scraping request, like `User-Agent`, `Accept-Language`, and `Content-Type`, what Crawlee sends by default, and how to change them. +Every request a crawler sends includes [HTTP headers](https://developer.mozilla.org/en-US/docs/Web/HTTP/Reference/Headers). These headers tell the server who is making the request, what content is acceptable, and in what language. The server reads them and decides what to return. The same URL can return different content, a different status code, or a blocked page depending on the headers it sees. This page covers the headers that shape a scraping request, like `User-Agent`, `Accept-Language`, and `Content-Type`, what Crawlee sends by default, and how to change them. ## What headers do @@ -44,7 +44,7 @@ These headers tell the server what the client can handle and the server uses the `Cookie` carries session and login state. Crawlee manages cookies through [sessions](./session-management), so you rarely set this one by hand. For details, see [Cookie management](./cookie-management). -`Authorization` carries credentials, such as a bearer token or basic auth. APIs commonly require it. Set it on the request when the target needs authenticated access. Treat its value as a secret, and don't send it through a [proxy you don't control](./security-of-web-scraping#untrusted-proxies). +`Authorization` carries credentials, such as a bearer token or basic auth. APIs commonly require it. Set it on the request when the target needs authenticated access. Treat its value as a secret, and don't send it through a [proxy you don't control](../guides/security-of-web-scraping#untrusted-proxies). ### Client hints and fingerprinting headers @@ -68,7 +68,7 @@ Each client implements impersonation its own way: - `HttpxHttpClient` uses a `HeaderGenerator` to add `Accept`, `Accept-Language`, and `User-Agent`. - `CurlImpersonateHttpClient` impersonates Chrome at the TLS and HTTP layer through [`curl-cffi`](https://curl-cffi.readthedocs.io/). -The header values match a specific version of a real browser, so the whole set stays internally consistent rather than a mix that no real client would send. For more on staying unblocked, see the [avoid blocking](./avoid-blocking) guide. +The header values match a specific version of a real browser, so the whole set stays internally consistent rather than a mix that no real client would send. For more on staying unblocked, see the [avoid blocking](../guides/avoid-blocking) guide. ## When impersonation hurts @@ -131,8 +131,3 @@ Anti-bot systems look at more than header values. They look at which headers are This fingerprinting is why `ImpitHttpClient` and `CurlImpersonateHttpClient` replicate the browser at the transport layer rather than just attaching headers. Setting a browser `User-Agent` on a plain client isn't enough to pass these checks. If a target uses fingerprinting, prefer an impersonating client over hand-set headers. -## Conclusion - -Headers decide what a server sends back. Crawlee impersonates a browser by default, which keeps a crawl unblocked on normal pages but can break endpoints that expect different headers. Turn impersonation off by building the client without it when you target such an endpoint, set custom headers on the client or per request, and reach for an impersonating client when the target fingerprints its traffic. - -If you have questions or need assistance, feel free to reach out on our [GitHub](https://github.com/apify/crawlee-python) or join our [Discord community](https://discord.com/invite/jyEM2PRvMU). Happy scraping! diff --git a/docs/guides/proxy_management.mdx b/docs/03_concepts/13_proxy_management.mdx similarity index 98% rename from docs/guides/proxy_management.mdx rename to docs/03_concepts/13_proxy_management.mdx index 38385ac950..7b6d74a627 100644 --- a/docs/guides/proxy_management.mdx +++ b/docs/03_concepts/13_proxy_management.mdx @@ -1,7 +1,7 @@ --- id: proxy-management title: Proxy management -description: Using proxies to get around those annoying IP-blocks +description: How to use and rotate proxies in Crawlee - proxy configuration, IP rotation tied to sessions, and tiered proxies. --- import ApiLink from '@site/src/components/ApiLink'; diff --git a/docs/guides/scaling_crawlers.mdx b/docs/03_concepts/14_scaling_crawlers.mdx similarity index 93% rename from docs/guides/scaling_crawlers.mdx rename to docs/03_concepts/14_scaling_crawlers.mdx index 152d852e60..0197b08294 100644 --- a/docs/guides/scaling_crawlers.mdx +++ b/docs/03_concepts/14_scaling_crawlers.mdx @@ -14,7 +14,7 @@ As we build our crawler, we may want to control how many tasks it performs at an :::tip -All of these options are available across all crawlers provided by Crawlee. In this guide, we are using the `BeautifulSoupCrawler` as an example. You should also explore the `ConcurrencySettings`. +All of these options are available across all crawlers provided by Crawlee. The examples on this page use the `BeautifulSoupCrawler`. You should also explore the `ConcurrencySettings`. ::: diff --git a/docs/guides/request_throttling.mdx b/docs/03_concepts/15_request_throttling.mdx similarity index 100% rename from docs/guides/request_throttling.mdx rename to docs/03_concepts/15_request_throttling.mdx diff --git a/docs/guides/error_handling.mdx b/docs/03_concepts/16_error_handling.mdx similarity index 56% rename from docs/guides/error_handling.mdx rename to docs/03_concepts/16_error_handling.mdx index abd1b33058..99d480d0ea 100644 --- a/docs/guides/error_handling.mdx +++ b/docs/03_concepts/16_error_handling.mdx @@ -5,13 +5,17 @@ description: How to handle errors that occur during web crawling. --- import ApiLink from '@site/src/components/ApiLink'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; import HandleProxyError from '!!raw-loader!roa-loader!./code_examples/error_handling/handle_proxy_error.py'; import ChangeHandleErrorStatus from '!!raw-loader!roa-loader!./code_examples/error_handling/change_handle_error_status.py'; import DisableRetry from '!!raw-loader!roa-loader!./code_examples/error_handling/disable_retry.py'; +import ParselCrawlerWithErrorSnapshotter from '!!raw-loader!roa-loader!./code_examples/error_handling/parsel_crawler_with_error_snapshotter.py'; +import PlaywrightCrawlerWithErrorSnapshotter from '!!raw-loader!roa-loader!./code_examples/error_handling/playwright_crawler_with_error_snapshotter.py'; -This guide demonstrates techniques for handling common errors encountered during web crawling operations. +This page demonstrates how Crawlee deals with errors during crawling and the techniques available for handling them. ## Handling proxy errors @@ -21,11 +25,11 @@ Low-quality proxies can cause problems even with high settings for `max_request_ {HandleProxyError} -You can use this same approach when testing different proxy providers. To better manage this process, you can count proxy errors and [stop the crawler](../examples/crawler-stop) if you get too many. +You can use this same approach when testing different proxy providers. To better manage this process, you can count proxy errors and [stop the crawler](../guides/stopping-and-resuming-crawlers#stopping-a-crawler) if you get too many. ## Changing how error status codes are handled -By default, when `Sessions` get status codes like [401](https://developer.mozilla.org/en-US/docs/Web/HTTP/Reference/Status/401), [403](https://developer.mozilla.org/en-US/docs/Web/HTTP/Reference/Status/403), or [429](https://developer.mozilla.org/en-US/docs/Web/HTTP/Reference/Status/429), Crawlee marks the `Session` as `retire` and switches to a new one. This might not be what you want, especially when working with [authentication](./logging-in-with-a-crawler). You can learn more in the [Session management guide](./session-management). +By default, when `Sessions` get status codes like [401](https://developer.mozilla.org/en-US/docs/Web/HTTP/Reference/Status/401), [403](https://developer.mozilla.org/en-US/docs/Web/HTTP/Reference/Status/403), or [429](https://developer.mozilla.org/en-US/docs/Web/HTTP/Reference/Status/429), Crawlee marks the `Session` as `retire` and switches to a new one. This might not be what you want, especially when working with [authentication](../guides/logging-in-with-a-crawler). You can learn more on the [Session management](./session-management) page. Here's an example of how to change this behavior: @@ -42,3 +46,20 @@ Here's how to turn off retries for non-network errors using {DisableRetry} + +## Capturing page snapshots on errors + +Debugging a failing request is much easier when you can see what the page looked like at the moment of failure. Set `save_error_snapshots=True` in the crawler's `Statistics` and Crawlee automatically captures a page snapshot on the first occurrence of each unique error. The snapshot can contain an HTML file and a JPEG screenshot of the page where the unhandled exception was raised, and it's saved to the default key-value store. Both `PlaywrightCrawler` and [HTTP crawlers](./http-crawlers) capture the HTML file, but only `PlaywrightCrawler` can capture a screenshot as well. + + + + + {ParselCrawlerWithErrorSnapshotter} + + + + + {PlaywrightCrawlerWithErrorSnapshotter} + + + diff --git a/docs/03_concepts/17_logging.mdx b/docs/03_concepts/17_logging.mdx new file mode 100644 index 0000000000..22fae255ea --- /dev/null +++ b/docs/03_concepts/17_logging.mdx @@ -0,0 +1,70 @@ +--- +id: logging +title: Logging +description: How Crawlee logs its activity and how to configure the log level, statistics logs, and structured JSON output. +--- + +import ApiLink from '@site/src/components/ApiLink'; +import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; + +import JsonLoggingExample from '!!raw-loader!roa-loader!./code_examples/logging/configure_json_logging.py'; + +Crawlee builds on the standard Python [`logging`](https://docs.python.org/3/library/logging.html) module. Each crawler comes with a logger. HTTP crawlers name it after the crawler class (for example `HttpCrawler`). Inside a request handler the same logger is available as `context.log`, so you can log from your handlers without any setup. + +## Default configuration + +By default, a crawler sets up the logging infrastructure automatically when it's created. It attaches a handler with a human-readable, colored formatter to the root logger, unless the root logger already has handlers configured. You can opt out by passing `configure_logging=False` to the crawler's constructor (see `BasicCrawlerOptions`) and configure logging yourself. + +The log level is controlled by the `Configuration.log_level` option, which can also be set through the `CRAWLEE_LOG_LEVEL` environment variable. + +## Statistics logs + +While running, the crawler periodically logs its statistics - requests finished and failed, retry histogram, durations, and more - and prints a final summary when the crawl ends. The `statistics_log_format` option of the crawler's constructor controls the output format: `table` (the default) renders the statistics as a formatted table, while `inline` logs them as a plain message with the values attached as extra fields, which is easier to consume with external log tooling. + +## JSON logging with loguru + +For log collection and analysis platforms such as the ELK Stack or Grafana Loki, line-delimited JSON (JSONL) output works better than formatted tables. The following example integrates Crawlee with the popular [`loguru`](https://github.com/delgan/loguru) library: an interceptor forwards all standard `logging` records to loguru, which serializes them as one JSON object per line. Note the crawler is created with `configure_logging=False` and `statistics_log_format='inline'`, so the statistics arrive as structured fields instead of a table. + + + {JsonLoggingExample} + + +Here's an example of what a crawler statistics log entry looks like in JSONL format: + +```json +{ + "text": "[HttpCrawler] | INFO | - Final request statistics: {'requests_finished': 1, 'requests_failed': 0, 'retry_histogram': [1], 'request_avg_failed_duration': None, 'request_avg_finished_duration': 3.57098, 'requests_finished_per_minute': 17, 'requests_failed_per_minute': 0, 'request_total_duration': 3.57098, 'requests_total': 1, 'crawler_runtime': 3.59165}\n", + "record": { + "elapsed": { "repr": "0:00:05.604568", "seconds": 5.604568 }, + "exception": null, + "extra": { + "requests_finished": 1, + "requests_failed": 0, + "retry_histogram": [1], + "request_avg_failed_duration": null, + "request_avg_finished_duration": 3.57098, + "requests_finished_per_minute": 17, + "requests_failed_per_minute": 0, + "request_total_duration": 3.57098, + "requests_total": 1, + "crawler_runtime": 3.59165 + }, + "file": { + "name": "_basic_crawler.py", + "path": "/crawlers/_basic/_basic_crawler.py" + }, + "function": "run", + "level": { "icon": "ℹ️", "name": "INFO", "no": 20 }, + "line": 583, + "message": "Final request statistics:", + "module": "_basic_crawler", + "name": "HttpCrawler", + "process": { "id": 198383, "name": "MainProcess" }, + "thread": { "id": 135312814966592, "name": "MainThread" }, + "time": { + "repr": "2025-03-17 17:14:45.339150+00:00", + "timestamp": 1742231685.33915 + } + } +} +``` diff --git a/docs/guides/service_locator.mdx b/docs/03_concepts/18_service_locator.mdx similarity index 91% rename from docs/guides/service_locator.mdx rename to docs/03_concepts/18_service_locator.mdx index fe10ce50c2..1f3d785111 100644 --- a/docs/guides/service_locator.mdx +++ b/docs/03_concepts/18_service_locator.mdx @@ -39,7 +39,7 @@ There are three core services that are managed by the service locator: `StorageClient` is the backend implementation for storages in Crawlee. It provides a unified interface for `Dataset`, `KeyValueStore`, and `RequestQueue`, regardless of the underlying storage implementation. Storage clients were already explained in the storage clients section. -Refer to the [Storage clients guide](./storage-clients) for more information about storage clients and how to use them. +Refer to the [Storage clients](./storage-clients) page for more information about storage clients and how to use them. ### EventManager @@ -129,8 +129,3 @@ Once a service has been retrieved from the service locator, attempting to set a {ServiceConflicts} -## Conclusion - -The `ServiceLocator` is a tool for managing global services in Crawlee. It provides a consistent way to configure and access services throughout the framework, ensuring that all components have access to the same configuration and services. - -If you have questions or need assistance, feel free to reach out on our [GitHub](https://github.com/apify/crawlee-python) or join our [Discord community](https://discord.com/invite/jyEM2PRvMU). Happy scraping! diff --git a/docs/guides/code_examples/cookie_management/initial_cookies.py b/docs/03_concepts/code_examples/cookie_management/initial_cookies.py similarity index 100% rename from docs/guides/code_examples/cookie_management/initial_cookies.py rename to docs/03_concepts/code_examples/cookie_management/initial_cookies.py diff --git a/docs/guides/code_examples/cookie_management/persist_cookies.py b/docs/03_concepts/code_examples/cookie_management/persist_cookies.py similarity index 100% rename from docs/guides/code_examples/cookie_management/persist_cookies.py rename to docs/03_concepts/code_examples/cookie_management/persist_cookies.py diff --git a/docs/guides/code_examples/cookie_management/playwright_cookies.py b/docs/03_concepts/code_examples/cookie_management/playwright_cookies.py similarity index 100% rename from docs/guides/code_examples/cookie_management/playwright_cookies.py rename to docs/03_concepts/code_examples/cookie_management/playwright_cookies.py diff --git a/docs/guides/code_examples/cookie_management/read_write_cookies.py b/docs/03_concepts/code_examples/cookie_management/read_write_cookies.py similarity index 100% rename from docs/guides/code_examples/cookie_management/read_write_cookies.py rename to docs/03_concepts/code_examples/cookie_management/read_write_cookies.py diff --git a/docs/guides/code_examples/cookie_management/retry_pinned_session.py b/docs/03_concepts/code_examples/cookie_management/retry_pinned_session.py similarity index 100% rename from docs/guides/code_examples/cookie_management/retry_pinned_session.py rename to docs/03_concepts/code_examples/cookie_management/retry_pinned_session.py diff --git a/docs/guides/code_examples/cookie_management/retry_restore_cookies.py b/docs/03_concepts/code_examples/cookie_management/retry_restore_cookies.py similarity index 100% rename from docs/guides/code_examples/cookie_management/retry_restore_cookies.py rename to docs/03_concepts/code_examples/cookie_management/retry_restore_cookies.py diff --git a/docs/guides/code_examples/cookie_management/retry_single_session.py b/docs/03_concepts/code_examples/cookie_management/retry_single_session.py similarity index 100% rename from docs/guides/code_examples/cookie_management/retry_single_session.py rename to docs/03_concepts/code_examples/cookie_management/retry_single_session.py diff --git a/docs/guides/code_examples/error_handling/change_handle_error_status.py b/docs/03_concepts/code_examples/error_handling/change_handle_error_status.py similarity index 100% rename from docs/guides/code_examples/error_handling/change_handle_error_status.py rename to docs/03_concepts/code_examples/error_handling/change_handle_error_status.py diff --git a/docs/guides/code_examples/error_handling/disable_retry.py b/docs/03_concepts/code_examples/error_handling/disable_retry.py similarity index 100% rename from docs/guides/code_examples/error_handling/disable_retry.py rename to docs/03_concepts/code_examples/error_handling/disable_retry.py diff --git a/docs/guides/code_examples/error_handling/handle_proxy_error.py b/docs/03_concepts/code_examples/error_handling/handle_proxy_error.py similarity index 100% rename from docs/guides/code_examples/error_handling/handle_proxy_error.py rename to docs/03_concepts/code_examples/error_handling/handle_proxy_error.py diff --git a/docs/examples/code_examples/parsel_crawler_with_error_snapshotter.py b/docs/03_concepts/code_examples/error_handling/parsel_crawler_with_error_snapshotter.py similarity index 100% rename from docs/examples/code_examples/parsel_crawler_with_error_snapshotter.py rename to docs/03_concepts/code_examples/error_handling/parsel_crawler_with_error_snapshotter.py diff --git a/docs/examples/code_examples/playwright_crawler_with_error_snapshotter.py b/docs/03_concepts/code_examples/error_handling/playwright_crawler_with_error_snapshotter.py similarity index 100% rename from docs/examples/code_examples/playwright_crawler_with_error_snapshotter.py rename to docs/03_concepts/code_examples/error_handling/playwright_crawler_with_error_snapshotter.py diff --git a/docs/guides/code_examples/http_clients/parsel_curl_impersonate_example.py b/docs/03_concepts/code_examples/http_clients/parsel_curl_impersonate_example.py similarity index 100% rename from docs/guides/code_examples/http_clients/parsel_curl_impersonate_example.py rename to docs/03_concepts/code_examples/http_clients/parsel_curl_impersonate_example.py diff --git a/docs/guides/code_examples/http_clients/parsel_httpx_example.py b/docs/03_concepts/code_examples/http_clients/parsel_httpx_example.py similarity index 100% rename from docs/guides/code_examples/http_clients/parsel_httpx_example.py rename to docs/03_concepts/code_examples/http_clients/parsel_httpx_example.py diff --git a/docs/guides/code_examples/http_clients/parsel_impit_example.py b/docs/03_concepts/code_examples/http_clients/parsel_impit_example.py similarity index 100% rename from docs/guides/code_examples/http_clients/parsel_impit_example.py rename to docs/03_concepts/code_examples/http_clients/parsel_impit_example.py diff --git a/docs/guides/code_examples/running_in_web_server/__init__.py b/docs/03_concepts/code_examples/http_crawlers/__init__.py similarity index 100% rename from docs/guides/code_examples/running_in_web_server/__init__.py rename to docs/03_concepts/code_examples/http_crawlers/__init__.py diff --git a/docs/guides/code_examples/http_crawlers/beautifulsoup_example.py b/docs/03_concepts/code_examples/http_crawlers/beautifulsoup_example.py similarity index 100% rename from docs/guides/code_examples/http_crawlers/beautifulsoup_example.py rename to docs/03_concepts/code_examples/http_crawlers/beautifulsoup_example.py diff --git a/docs/guides/code_examples/http_crawlers/custom_crawler_example.py b/docs/03_concepts/code_examples/http_crawlers/custom_crawler_example.py similarity index 100% rename from docs/guides/code_examples/http_crawlers/custom_crawler_example.py rename to docs/03_concepts/code_examples/http_crawlers/custom_crawler_example.py diff --git a/docs/examples/code_examples/file_download.py b/docs/03_concepts/code_examples/http_crawlers/file_download.py similarity index 100% rename from docs/examples/code_examples/file_download.py rename to docs/03_concepts/code_examples/http_crawlers/file_download.py diff --git a/docs/examples/code_examples/file_download_stream.py b/docs/03_concepts/code_examples/http_crawlers/file_download_stream.py similarity index 100% rename from docs/examples/code_examples/file_download_stream.py rename to docs/03_concepts/code_examples/http_crawlers/file_download_stream.py diff --git a/docs/guides/code_examples/http_crawlers/http_example.py b/docs/03_concepts/code_examples/http_crawlers/http_example.py similarity index 100% rename from docs/guides/code_examples/http_crawlers/http_example.py rename to docs/03_concepts/code_examples/http_crawlers/http_example.py diff --git a/docs/guides/code_examples/http_crawlers/lexbor_parser.py b/docs/03_concepts/code_examples/http_crawlers/lexbor_parser.py similarity index 100% rename from docs/guides/code_examples/http_crawlers/lexbor_parser.py rename to docs/03_concepts/code_examples/http_crawlers/lexbor_parser.py diff --git a/docs/guides/code_examples/http_crawlers/lxml_parser.py b/docs/03_concepts/code_examples/http_crawlers/lxml_parser.py similarity index 100% rename from docs/guides/code_examples/http_crawlers/lxml_parser.py rename to docs/03_concepts/code_examples/http_crawlers/lxml_parser.py diff --git a/docs/guides/code_examples/http_crawlers/lxml_saxonche_parser.py b/docs/03_concepts/code_examples/http_crawlers/lxml_saxonche_parser.py similarity index 100% rename from docs/guides/code_examples/http_crawlers/lxml_saxonche_parser.py rename to docs/03_concepts/code_examples/http_crawlers/lxml_saxonche_parser.py diff --git a/docs/guides/code_examples/http_crawlers/parsel_example.py b/docs/03_concepts/code_examples/http_crawlers/parsel_example.py similarity index 100% rename from docs/guides/code_examples/http_crawlers/parsel_example.py rename to docs/03_concepts/code_examples/http_crawlers/parsel_example.py diff --git a/docs/guides/code_examples/http_crawlers/pyquery_parser.py b/docs/03_concepts/code_examples/http_crawlers/pyquery_parser.py similarity index 100% rename from docs/guides/code_examples/http_crawlers/pyquery_parser.py rename to docs/03_concepts/code_examples/http_crawlers/pyquery_parser.py diff --git a/docs/guides/code_examples/http_crawlers/scrapling_parser.py b/docs/03_concepts/code_examples/http_crawlers/scrapling_parser.py similarity index 100% rename from docs/guides/code_examples/http_crawlers/scrapling_parser.py rename to docs/03_concepts/code_examples/http_crawlers/scrapling_parser.py diff --git a/docs/guides/code_examples/http_crawlers/selectolax_adaptive_run.py b/docs/03_concepts/code_examples/http_crawlers/selectolax_adaptive_run.py similarity index 100% rename from docs/guides/code_examples/http_crawlers/selectolax_adaptive_run.py rename to docs/03_concepts/code_examples/http_crawlers/selectolax_adaptive_run.py diff --git a/docs/guides/code_examples/http_crawlers/selectolax_context.py b/docs/03_concepts/code_examples/http_crawlers/selectolax_context.py similarity index 100% rename from docs/guides/code_examples/http_crawlers/selectolax_context.py rename to docs/03_concepts/code_examples/http_crawlers/selectolax_context.py diff --git a/docs/guides/code_examples/http_crawlers/selectolax_crawler.py b/docs/03_concepts/code_examples/http_crawlers/selectolax_crawler.py similarity index 100% rename from docs/guides/code_examples/http_crawlers/selectolax_crawler.py rename to docs/03_concepts/code_examples/http_crawlers/selectolax_crawler.py diff --git a/docs/guides/code_examples/http_crawlers/selectolax_crawler_run.py b/docs/03_concepts/code_examples/http_crawlers/selectolax_crawler_run.py similarity index 100% rename from docs/guides/code_examples/http_crawlers/selectolax_crawler_run.py rename to docs/03_concepts/code_examples/http_crawlers/selectolax_crawler_run.py diff --git a/docs/guides/code_examples/http_crawlers/selectolax_parser.py b/docs/03_concepts/code_examples/http_crawlers/selectolax_parser.py similarity index 100% rename from docs/guides/code_examples/http_crawlers/selectolax_parser.py rename to docs/03_concepts/code_examples/http_crawlers/selectolax_parser.py diff --git a/docs/guides/code_examples/http_headers/browser_page_headers.py b/docs/03_concepts/code_examples/http_headers/browser_page_headers.py similarity index 100% rename from docs/guides/code_examples/http_headers/browser_page_headers.py rename to docs/03_concepts/code_examples/http_headers/browser_page_headers.py diff --git a/docs/guides/code_examples/http_headers/set_headers.py b/docs/03_concepts/code_examples/http_headers/set_headers.py similarity index 100% rename from docs/guides/code_examples/http_headers/set_headers.py rename to docs/03_concepts/code_examples/http_headers/set_headers.py diff --git a/docs/examples/code_examples/configure_json_logging.py b/docs/03_concepts/code_examples/logging/configure_json_logging.py similarity index 100% rename from docs/examples/code_examples/configure_json_logging.py rename to docs/03_concepts/code_examples/logging/configure_json_logging.py diff --git a/docs/guides/code_examples/playwright_crawler/browser_configuration_example.py b/docs/03_concepts/code_examples/playwright_crawler/browser_configuration_example.py similarity index 100% rename from docs/guides/code_examples/playwright_crawler/browser_configuration_example.py rename to docs/03_concepts/code_examples/playwright_crawler/browser_configuration_example.py diff --git a/docs/guides/code_examples/playwright_crawler/browser_pool_launch_hooks_example.py b/docs/03_concepts/code_examples/playwright_crawler/browser_pool_launch_hooks_example.py similarity index 100% rename from docs/guides/code_examples/playwright_crawler/browser_pool_launch_hooks_example.py rename to docs/03_concepts/code_examples/playwright_crawler/browser_pool_launch_hooks_example.py diff --git a/docs/guides/code_examples/playwright_crawler/browser_pool_page_hooks_example.py b/docs/03_concepts/code_examples/playwright_crawler/browser_pool_page_hooks_example.py similarity index 100% rename from docs/guides/code_examples/playwright_crawler/browser_pool_page_hooks_example.py rename to docs/03_concepts/code_examples/playwright_crawler/browser_pool_page_hooks_example.py diff --git a/docs/examples/code_examples/capture_screenshot_using_playwright.py b/docs/03_concepts/code_examples/playwright_crawler/capture_screenshot_using_playwright.py similarity index 100% rename from docs/examples/code_examples/capture_screenshot_using_playwright.py rename to docs/03_concepts/code_examples/playwright_crawler/capture_screenshot_using_playwright.py diff --git a/docs/guides/code_examples/playwright_crawler/multiple_launch_example.py b/docs/03_concepts/code_examples/playwright_crawler/multiple_launch_example.py similarity index 100% rename from docs/guides/code_examples/playwright_crawler/multiple_launch_example.py rename to docs/03_concepts/code_examples/playwright_crawler/multiple_launch_example.py diff --git a/docs/guides/code_examples/playwright_crawler/navigation_hooks_example.py b/docs/03_concepts/code_examples/playwright_crawler/navigation_hooks_example.py similarity index 100% rename from docs/guides/code_examples/playwright_crawler/navigation_hooks_example.py rename to docs/03_concepts/code_examples/playwright_crawler/navigation_hooks_example.py diff --git a/docs/examples/code_examples/playwright_block_requests.py b/docs/03_concepts/code_examples/playwright_crawler/playwright_block_requests.py similarity index 100% rename from docs/examples/code_examples/playwright_block_requests.py rename to docs/03_concepts/code_examples/playwright_crawler/playwright_block_requests.py diff --git a/docs/examples/code_examples/playwright_crawler.py b/docs/03_concepts/code_examples/playwright_crawler/playwright_crawler.py similarity index 100% rename from docs/examples/code_examples/playwright_crawler.py rename to docs/03_concepts/code_examples/playwright_crawler/playwright_crawler.py diff --git a/docs/guides/code_examples/playwright_crawler/plugin_browser_configuration_example.py b/docs/03_concepts/code_examples/playwright_crawler/plugin_browser_configuration_example.py similarity index 100% rename from docs/guides/code_examples/playwright_crawler/plugin_browser_configuration_example.py rename to docs/03_concepts/code_examples/playwright_crawler/plugin_browser_configuration_example.py diff --git a/docs/examples/code_examples/adaptive_playwright_crawler.py b/docs/03_concepts/code_examples/playwright_crawler_adaptive/adaptive_playwright_crawler.py similarity index 100% rename from docs/examples/code_examples/adaptive_playwright_crawler.py rename to docs/03_concepts/code_examples/playwright_crawler_adaptive/adaptive_playwright_crawler.py diff --git a/docs/guides/code_examples/playwright_crawler_adaptive/handler.py b/docs/03_concepts/code_examples/playwright_crawler_adaptive/handler.py similarity index 100% rename from docs/guides/code_examples/playwright_crawler_adaptive/handler.py rename to docs/03_concepts/code_examples/playwright_crawler_adaptive/handler.py diff --git a/docs/guides/code_examples/playwright_crawler_adaptive/init_beautifulsoup.py b/docs/03_concepts/code_examples/playwright_crawler_adaptive/init_beautifulsoup.py similarity index 100% rename from docs/guides/code_examples/playwright_crawler_adaptive/init_beautifulsoup.py rename to docs/03_concepts/code_examples/playwright_crawler_adaptive/init_beautifulsoup.py diff --git a/docs/guides/code_examples/playwright_crawler_adaptive/init_parsel.py b/docs/03_concepts/code_examples/playwright_crawler_adaptive/init_parsel.py similarity index 100% rename from docs/guides/code_examples/playwright_crawler_adaptive/init_parsel.py rename to docs/03_concepts/code_examples/playwright_crawler_adaptive/init_parsel.py diff --git a/docs/guides/code_examples/playwright_crawler_adaptive/init_prediction.py b/docs/03_concepts/code_examples/playwright_crawler_adaptive/init_prediction.py similarity index 100% rename from docs/guides/code_examples/playwright_crawler_adaptive/init_prediction.py rename to docs/03_concepts/code_examples/playwright_crawler_adaptive/init_prediction.py diff --git a/docs/guides/code_examples/playwright_crawler_adaptive/pre_nav_hooks.py b/docs/03_concepts/code_examples/playwright_crawler_adaptive/pre_nav_hooks.py similarity index 100% rename from docs/guides/code_examples/playwright_crawler_adaptive/pre_nav_hooks.py rename to docs/03_concepts/code_examples/playwright_crawler_adaptive/pre_nav_hooks.py diff --git a/docs/guides/code_examples/proxy_management/inspecting_bs_example.py b/docs/03_concepts/code_examples/proxy_management/inspecting_bs_example.py similarity index 100% rename from docs/guides/code_examples/proxy_management/inspecting_bs_example.py rename to docs/03_concepts/code_examples/proxy_management/inspecting_bs_example.py diff --git a/docs/guides/code_examples/proxy_management/inspecting_pw_example.py b/docs/03_concepts/code_examples/proxy_management/inspecting_pw_example.py similarity index 100% rename from docs/guides/code_examples/proxy_management/inspecting_pw_example.py rename to docs/03_concepts/code_examples/proxy_management/inspecting_pw_example.py diff --git a/docs/guides/code_examples/proxy_management/integration_bs_example.py b/docs/03_concepts/code_examples/proxy_management/integration_bs_example.py similarity index 100% rename from docs/guides/code_examples/proxy_management/integration_bs_example.py rename to docs/03_concepts/code_examples/proxy_management/integration_bs_example.py diff --git a/docs/guides/code_examples/proxy_management/integration_pw_example.py b/docs/03_concepts/code_examples/proxy_management/integration_pw_example.py similarity index 100% rename from docs/guides/code_examples/proxy_management/integration_pw_example.py rename to docs/03_concepts/code_examples/proxy_management/integration_pw_example.py diff --git a/docs/guides/code_examples/proxy_management/quick_start_example.py b/docs/03_concepts/code_examples/proxy_management/quick_start_example.py similarity index 100% rename from docs/guides/code_examples/proxy_management/quick_start_example.py rename to docs/03_concepts/code_examples/proxy_management/quick_start_example.py diff --git a/docs/guides/code_examples/proxy_management/session_bs_example.py b/docs/03_concepts/code_examples/proxy_management/session_bs_example.py similarity index 100% rename from docs/guides/code_examples/proxy_management/session_bs_example.py rename to docs/03_concepts/code_examples/proxy_management/session_bs_example.py diff --git a/docs/guides/code_examples/proxy_management/session_pw_example.py b/docs/03_concepts/code_examples/proxy_management/session_pw_example.py similarity index 100% rename from docs/guides/code_examples/proxy_management/session_pw_example.py rename to docs/03_concepts/code_examples/proxy_management/session_pw_example.py diff --git a/docs/guides/code_examples/proxy_management/tiers_bs_example.py b/docs/03_concepts/code_examples/proxy_management/tiers_bs_example.py similarity index 100% rename from docs/guides/code_examples/proxy_management/tiers_bs_example.py rename to docs/03_concepts/code_examples/proxy_management/tiers_bs_example.py diff --git a/docs/guides/code_examples/proxy_management/tiers_pw_example.py b/docs/03_concepts/code_examples/proxy_management/tiers_pw_example.py similarity index 100% rename from docs/guides/code_examples/proxy_management/tiers_pw_example.py rename to docs/03_concepts/code_examples/proxy_management/tiers_pw_example.py diff --git a/docs/guides/code_examples/request_loaders/rl_basic_example.py b/docs/03_concepts/code_examples/request_loaders/rl_basic_example.py similarity index 100% rename from docs/guides/code_examples/request_loaders/rl_basic_example.py rename to docs/03_concepts/code_examples/request_loaders/rl_basic_example.py diff --git a/docs/guides/code_examples/request_loaders/rl_basic_example_with_persist.py b/docs/03_concepts/code_examples/request_loaders/rl_basic_example_with_persist.py similarity index 100% rename from docs/guides/code_examples/request_loaders/rl_basic_example_with_persist.py rename to docs/03_concepts/code_examples/request_loaders/rl_basic_example_with_persist.py diff --git a/docs/guides/code_examples/request_loaders/rl_tandem_example.py b/docs/03_concepts/code_examples/request_loaders/rl_tandem_example.py similarity index 100% rename from docs/guides/code_examples/request_loaders/rl_tandem_example.py rename to docs/03_concepts/code_examples/request_loaders/rl_tandem_example.py diff --git a/docs/guides/code_examples/request_loaders/rl_tandem_example_explicit.py b/docs/03_concepts/code_examples/request_loaders/rl_tandem_example_explicit.py similarity index 100% rename from docs/guides/code_examples/request_loaders/rl_tandem_example_explicit.py rename to docs/03_concepts/code_examples/request_loaders/rl_tandem_example_explicit.py diff --git a/docs/guides/code_examples/request_loaders/sitemap_basic_example.py b/docs/03_concepts/code_examples/request_loaders/sitemap_basic_example.py similarity index 100% rename from docs/guides/code_examples/request_loaders/sitemap_basic_example.py rename to docs/03_concepts/code_examples/request_loaders/sitemap_basic_example.py diff --git a/docs/guides/code_examples/request_loaders/sitemap_example_with_persist.py b/docs/03_concepts/code_examples/request_loaders/sitemap_example_with_persist.py similarity index 100% rename from docs/guides/code_examples/request_loaders/sitemap_example_with_persist.py rename to docs/03_concepts/code_examples/request_loaders/sitemap_example_with_persist.py diff --git a/docs/guides/code_examples/request_loaders/sitemap_tandem_example.py b/docs/03_concepts/code_examples/request_loaders/sitemap_tandem_example.py similarity index 100% rename from docs/guides/code_examples/request_loaders/sitemap_tandem_example.py rename to docs/03_concepts/code_examples/request_loaders/sitemap_tandem_example.py diff --git a/docs/guides/code_examples/request_loaders/sitemap_tandem_example_explicit.py b/docs/03_concepts/code_examples/request_loaders/sitemap_tandem_example_explicit.py similarity index 100% rename from docs/guides/code_examples/request_loaders/sitemap_tandem_example_explicit.py rename to docs/03_concepts/code_examples/request_loaders/sitemap_tandem_example_explicit.py diff --git a/docs/examples/code_examples/using_sitemap_request_loader.py b/docs/03_concepts/code_examples/request_loaders/using_sitemap_request_loader.py similarity index 100% rename from docs/examples/code_examples/using_sitemap_request_loader.py rename to docs/03_concepts/code_examples/request_loaders/using_sitemap_request_loader.py diff --git a/docs/guides/code_examples/request_router/adaptive_crawler_handlers.py b/docs/03_concepts/code_examples/request_router/adaptive_crawler_handlers.py similarity index 100% rename from docs/guides/code_examples/request_router/adaptive_crawler_handlers.py rename to docs/03_concepts/code_examples/request_router/adaptive_crawler_handlers.py diff --git a/docs/guides/code_examples/request_router/basic_request_handlers.py b/docs/03_concepts/code_examples/request_router/basic_request_handlers.py similarity index 100% rename from docs/guides/code_examples/request_router/basic_request_handlers.py rename to docs/03_concepts/code_examples/request_router/basic_request_handlers.py diff --git a/docs/guides/code_examples/request_router/custom_router_default_only.py b/docs/03_concepts/code_examples/request_router/custom_router_default_only.py similarity index 100% rename from docs/guides/code_examples/request_router/custom_router_default_only.py rename to docs/03_concepts/code_examples/request_router/custom_router_default_only.py diff --git a/docs/guides/code_examples/request_router/error_handler.py b/docs/03_concepts/code_examples/request_router/error_handler.py similarity index 100% rename from docs/guides/code_examples/request_router/error_handler.py rename to docs/03_concepts/code_examples/request_router/error_handler.py diff --git a/docs/guides/code_examples/request_router/failed_request_handler.py b/docs/03_concepts/code_examples/request_router/failed_request_handler.py similarity index 100% rename from docs/guides/code_examples/request_router/failed_request_handler.py rename to docs/03_concepts/code_examples/request_router/failed_request_handler.py diff --git a/docs/guides/code_examples/request_router/http_pre_navigation.py b/docs/03_concepts/code_examples/request_router/http_pre_navigation.py similarity index 100% rename from docs/guides/code_examples/request_router/http_pre_navigation.py rename to docs/03_concepts/code_examples/request_router/http_pre_navigation.py diff --git a/docs/guides/code_examples/request_router/playwright_pre_navigation.py b/docs/03_concepts/code_examples/request_router/playwright_pre_navigation.py similarity index 100% rename from docs/guides/code_examples/request_router/playwright_pre_navigation.py rename to docs/03_concepts/code_examples/request_router/playwright_pre_navigation.py diff --git a/docs/guides/code_examples/request_router/router_middleware.py b/docs/03_concepts/code_examples/request_router/router_middleware.py similarity index 100% rename from docs/guides/code_examples/request_router/router_middleware.py rename to docs/03_concepts/code_examples/request_router/router_middleware.py diff --git a/docs/guides/code_examples/request_router/simple_default_handler.py b/docs/03_concepts/code_examples/request_router/simple_default_handler.py similarity index 100% rename from docs/guides/code_examples/request_router/simple_default_handler.py rename to docs/03_concepts/code_examples/request_router/simple_default_handler.py diff --git a/docs/guides/code_examples/request_throttling/throttling_example.py b/docs/03_concepts/code_examples/request_throttling/throttling_example.py similarity index 100% rename from docs/guides/code_examples/request_throttling/throttling_example.py rename to docs/03_concepts/code_examples/request_throttling/throttling_example.py diff --git a/docs/guides/code_examples/scaling_crawlers/max_tasks_per_minute_example.py b/docs/03_concepts/code_examples/scaling_crawlers/max_tasks_per_minute_example.py similarity index 100% rename from docs/guides/code_examples/scaling_crawlers/max_tasks_per_minute_example.py rename to docs/03_concepts/code_examples/scaling_crawlers/max_tasks_per_minute_example.py diff --git a/docs/guides/code_examples/scaling_crawlers/min_and_max_concurrency_example.py b/docs/03_concepts/code_examples/scaling_crawlers/min_and_max_concurrency_example.py similarity index 100% rename from docs/guides/code_examples/scaling_crawlers/min_and_max_concurrency_example.py rename to docs/03_concepts/code_examples/scaling_crawlers/min_and_max_concurrency_example.py diff --git a/docs/guides/code_examples/service_locator/service_conflicts.py b/docs/03_concepts/code_examples/service_locator/service_conflicts.py similarity index 100% rename from docs/guides/code_examples/service_locator/service_conflicts.py rename to docs/03_concepts/code_examples/service_locator/service_conflicts.py diff --git a/docs/guides/code_examples/service_locator/service_crawler_configuration.py b/docs/03_concepts/code_examples/service_locator/service_crawler_configuration.py similarity index 100% rename from docs/guides/code_examples/service_locator/service_crawler_configuration.py rename to docs/03_concepts/code_examples/service_locator/service_crawler_configuration.py diff --git a/docs/guides/code_examples/service_locator/service_crawler_event_manager.py b/docs/03_concepts/code_examples/service_locator/service_crawler_event_manager.py similarity index 100% rename from docs/guides/code_examples/service_locator/service_crawler_event_manager.py rename to docs/03_concepts/code_examples/service_locator/service_crawler_event_manager.py diff --git a/docs/guides/code_examples/service_locator/service_crawler_storage_client.py b/docs/03_concepts/code_examples/service_locator/service_crawler_storage_client.py similarity index 100% rename from docs/guides/code_examples/service_locator/service_crawler_storage_client.py rename to docs/03_concepts/code_examples/service_locator/service_crawler_storage_client.py diff --git a/docs/guides/code_examples/service_locator/service_locator_configuration.py b/docs/03_concepts/code_examples/service_locator/service_locator_configuration.py similarity index 100% rename from docs/guides/code_examples/service_locator/service_locator_configuration.py rename to docs/03_concepts/code_examples/service_locator/service_locator_configuration.py diff --git a/docs/guides/code_examples/service_locator/service_locator_event_manager.py b/docs/03_concepts/code_examples/service_locator/service_locator_event_manager.py similarity index 100% rename from docs/guides/code_examples/service_locator/service_locator_event_manager.py rename to docs/03_concepts/code_examples/service_locator/service_locator_event_manager.py diff --git a/docs/guides/code_examples/service_locator/service_locator_storage_client.py b/docs/03_concepts/code_examples/service_locator/service_locator_storage_client.py similarity index 100% rename from docs/guides/code_examples/service_locator/service_locator_storage_client.py rename to docs/03_concepts/code_examples/service_locator/service_locator_storage_client.py diff --git a/docs/guides/code_examples/service_locator/service_storage_configuration.py b/docs/03_concepts/code_examples/service_locator/service_storage_configuration.py similarity index 100% rename from docs/guides/code_examples/service_locator/service_storage_configuration.py rename to docs/03_concepts/code_examples/service_locator/service_storage_configuration.py diff --git a/docs/guides/code_examples/service_locator/service_storage_storage_client.py b/docs/03_concepts/code_examples/service_locator/service_storage_storage_client.py similarity index 100% rename from docs/guides/code_examples/service_locator/service_storage_storage_client.py rename to docs/03_concepts/code_examples/service_locator/service_storage_storage_client.py diff --git a/docs/guides/code_examples/session_management/multi_sessions_http.py b/docs/03_concepts/code_examples/session_management/multi_sessions_http.py similarity index 100% rename from docs/guides/code_examples/session_management/multi_sessions_http.py rename to docs/03_concepts/code_examples/session_management/multi_sessions_http.py diff --git a/docs/guides/code_examples/session_management/one_session_http.py b/docs/03_concepts/code_examples/session_management/one_session_http.py similarity index 100% rename from docs/guides/code_examples/session_management/one_session_http.py rename to docs/03_concepts/code_examples/session_management/one_session_http.py diff --git a/docs/guides/code_examples/session_management/sm_basic.py b/docs/03_concepts/code_examples/session_management/sm_basic.py similarity index 100% rename from docs/guides/code_examples/session_management/sm_basic.py rename to docs/03_concepts/code_examples/session_management/sm_basic.py diff --git a/docs/guides/code_examples/session_management/sm_beautifulsoup.py b/docs/03_concepts/code_examples/session_management/sm_beautifulsoup.py similarity index 100% rename from docs/guides/code_examples/session_management/sm_beautifulsoup.py rename to docs/03_concepts/code_examples/session_management/sm_beautifulsoup.py diff --git a/docs/guides/code_examples/session_management/sm_http.py b/docs/03_concepts/code_examples/session_management/sm_http.py similarity index 100% rename from docs/guides/code_examples/session_management/sm_http.py rename to docs/03_concepts/code_examples/session_management/sm_http.py diff --git a/docs/guides/code_examples/session_management/sm_parsel.py b/docs/03_concepts/code_examples/session_management/sm_parsel.py similarity index 100% rename from docs/guides/code_examples/session_management/sm_parsel.py rename to docs/03_concepts/code_examples/session_management/sm_parsel.py diff --git a/docs/guides/code_examples/session_management/sm_playwright.py b/docs/03_concepts/code_examples/session_management/sm_playwright.py similarity index 100% rename from docs/guides/code_examples/session_management/sm_playwright.py rename to docs/03_concepts/code_examples/session_management/sm_playwright.py diff --git a/docs/guides/code_examples/session_management/sm_standalone.py b/docs/03_concepts/code_examples/session_management/sm_standalone.py similarity index 100% rename from docs/guides/code_examples/session_management/sm_standalone.py rename to docs/03_concepts/code_examples/session_management/sm_standalone.py diff --git a/docs/guides/code_examples/storage_clients/custom_storage_client_example.py b/docs/03_concepts/code_examples/storage_clients/custom_storage_client_example.py similarity index 100% rename from docs/guides/code_examples/storage_clients/custom_storage_client_example.py rename to docs/03_concepts/code_examples/storage_clients/custom_storage_client_example.py diff --git a/docs/guides/code_examples/storage_clients/file_system_storage_client_basic_example.py b/docs/03_concepts/code_examples/storage_clients/file_system_storage_client_basic_example.py similarity index 100% rename from docs/guides/code_examples/storage_clients/file_system_storage_client_basic_example.py rename to docs/03_concepts/code_examples/storage_clients/file_system_storage_client_basic_example.py diff --git a/docs/guides/code_examples/storage_clients/file_system_storage_client_configuration_example.py b/docs/03_concepts/code_examples/storage_clients/file_system_storage_client_configuration_example.py similarity index 100% rename from docs/guides/code_examples/storage_clients/file_system_storage_client_configuration_example.py rename to docs/03_concepts/code_examples/storage_clients/file_system_storage_client_configuration_example.py diff --git a/docs/guides/code_examples/storage_clients/memory_storage_client_basic_example.py b/docs/03_concepts/code_examples/storage_clients/memory_storage_client_basic_example.py similarity index 100% rename from docs/guides/code_examples/storage_clients/memory_storage_client_basic_example.py rename to docs/03_concepts/code_examples/storage_clients/memory_storage_client_basic_example.py diff --git a/docs/guides/code_examples/storage_clients/redis_storage_client_basic_example.py b/docs/03_concepts/code_examples/storage_clients/redis_storage_client_basic_example.py similarity index 100% rename from docs/guides/code_examples/storage_clients/redis_storage_client_basic_example.py rename to docs/03_concepts/code_examples/storage_clients/redis_storage_client_basic_example.py diff --git a/docs/guides/code_examples/storage_clients/redis_storage_client_configuration_example.py b/docs/03_concepts/code_examples/storage_clients/redis_storage_client_configuration_example.py similarity index 100% rename from docs/guides/code_examples/storage_clients/redis_storage_client_configuration_example.py rename to docs/03_concepts/code_examples/storage_clients/redis_storage_client_configuration_example.py diff --git a/docs/guides/code_examples/storage_clients/registering_storage_clients_example.py b/docs/03_concepts/code_examples/storage_clients/registering_storage_clients_example.py similarity index 100% rename from docs/guides/code_examples/storage_clients/registering_storage_clients_example.py rename to docs/03_concepts/code_examples/storage_clients/registering_storage_clients_example.py diff --git a/docs/guides/code_examples/storage_clients/sql_storage_client_basic_example.py b/docs/03_concepts/code_examples/storage_clients/sql_storage_client_basic_example.py similarity index 100% rename from docs/guides/code_examples/storage_clients/sql_storage_client_basic_example.py rename to docs/03_concepts/code_examples/storage_clients/sql_storage_client_basic_example.py diff --git a/docs/guides/code_examples/storage_clients/sql_storage_client_configuration_example.py b/docs/03_concepts/code_examples/storage_clients/sql_storage_client_configuration_example.py similarity index 100% rename from docs/guides/code_examples/storage_clients/sql_storage_client_configuration_example.py rename to docs/03_concepts/code_examples/storage_clients/sql_storage_client_configuration_example.py diff --git a/docs/guides/code_examples/storages/cleaning_do_not_purge_example.py b/docs/03_concepts/code_examples/storages/cleaning_do_not_purge_example.py similarity index 100% rename from docs/guides/code_examples/storages/cleaning_do_not_purge_example.py rename to docs/03_concepts/code_examples/storages/cleaning_do_not_purge_example.py diff --git a/docs/guides/code_examples/storages/cleaning_purge_explicitly_example.py b/docs/03_concepts/code_examples/storages/cleaning_purge_explicitly_example.py similarity index 100% rename from docs/guides/code_examples/storages/cleaning_purge_explicitly_example.py rename to docs/03_concepts/code_examples/storages/cleaning_purge_explicitly_example.py diff --git a/docs/guides/code_examples/storages/dataset_basic_example.py b/docs/03_concepts/code_examples/storages/dataset_basic_example.py similarity index 100% rename from docs/guides/code_examples/storages/dataset_basic_example.py rename to docs/03_concepts/code_examples/storages/dataset_basic_example.py diff --git a/docs/guides/code_examples/storages/dataset_with_crawler_example.py b/docs/03_concepts/code_examples/storages/dataset_with_crawler_example.py similarity index 100% rename from docs/guides/code_examples/storages/dataset_with_crawler_example.py rename to docs/03_concepts/code_examples/storages/dataset_with_crawler_example.py diff --git a/docs/guides/code_examples/storages/dataset_with_crawler_explicit_example.py b/docs/03_concepts/code_examples/storages/dataset_with_crawler_explicit_example.py similarity index 100% rename from docs/guides/code_examples/storages/dataset_with_crawler_explicit_example.py rename to docs/03_concepts/code_examples/storages/dataset_with_crawler_explicit_example.py diff --git a/docs/examples/code_examples/export_entire_dataset_to_file_csv.py b/docs/03_concepts/code_examples/storages/export_entire_dataset_to_file_csv.py similarity index 100% rename from docs/examples/code_examples/export_entire_dataset_to_file_csv.py rename to docs/03_concepts/code_examples/storages/export_entire_dataset_to_file_csv.py diff --git a/docs/examples/code_examples/export_entire_dataset_to_file_json.py b/docs/03_concepts/code_examples/storages/export_entire_dataset_to_file_json.py similarity index 100% rename from docs/examples/code_examples/export_entire_dataset_to_file_json.py rename to docs/03_concepts/code_examples/storages/export_entire_dataset_to_file_json.py diff --git a/docs/guides/code_examples/storages/helper_add_requests_example.py b/docs/03_concepts/code_examples/storages/helper_add_requests_example.py similarity index 100% rename from docs/guides/code_examples/storages/helper_add_requests_example.py rename to docs/03_concepts/code_examples/storages/helper_add_requests_example.py diff --git a/docs/guides/code_examples/storages/helper_enqueue_links_example.py b/docs/03_concepts/code_examples/storages/helper_enqueue_links_example.py similarity index 100% rename from docs/guides/code_examples/storages/helper_enqueue_links_example.py rename to docs/03_concepts/code_examples/storages/helper_enqueue_links_example.py diff --git a/docs/guides/code_examples/storages/kvs_basic_example.py b/docs/03_concepts/code_examples/storages/kvs_basic_example.py similarity index 100% rename from docs/guides/code_examples/storages/kvs_basic_example.py rename to docs/03_concepts/code_examples/storages/kvs_basic_example.py diff --git a/docs/guides/code_examples/storages/kvs_with_crawler_example.py b/docs/03_concepts/code_examples/storages/kvs_with_crawler_example.py similarity index 100% rename from docs/guides/code_examples/storages/kvs_with_crawler_example.py rename to docs/03_concepts/code_examples/storages/kvs_with_crawler_example.py diff --git a/docs/guides/code_examples/storages/kvs_with_crawler_explicit_example.py b/docs/03_concepts/code_examples/storages/kvs_with_crawler_explicit_example.py similarity index 100% rename from docs/guides/code_examples/storages/kvs_with_crawler_explicit_example.py rename to docs/03_concepts/code_examples/storages/kvs_with_crawler_explicit_example.py diff --git a/docs/guides/code_examples/storages/opening.py b/docs/03_concepts/code_examples/storages/opening.py similarity index 100% rename from docs/guides/code_examples/storages/opening.py rename to docs/03_concepts/code_examples/storages/opening.py diff --git a/docs/guides/code_examples/storages/rq_basic_example.py b/docs/03_concepts/code_examples/storages/rq_basic_example.py similarity index 100% rename from docs/guides/code_examples/storages/rq_basic_example.py rename to docs/03_concepts/code_examples/storages/rq_basic_example.py diff --git a/docs/guides/code_examples/storages/rq_with_crawler_example.py b/docs/03_concepts/code_examples/storages/rq_with_crawler_example.py similarity index 100% rename from docs/guides/code_examples/storages/rq_with_crawler_example.py rename to docs/03_concepts/code_examples/storages/rq_with_crawler_example.py diff --git a/docs/guides/code_examples/storages/rq_with_crawler_explicit_example.py b/docs/03_concepts/code_examples/storages/rq_with_crawler_explicit_example.py similarity index 100% rename from docs/guides/code_examples/storages/rq_with_crawler_explicit_example.py rename to docs/03_concepts/code_examples/storages/rq_with_crawler_explicit_example.py diff --git a/docs/04_guides/01_crawling_links.mdx b/docs/04_guides/01_crawling_links.mdx new file mode 100644 index 0000000000..bc98f501f1 --- /dev/null +++ b/docs/04_guides/01_crawling_links.mdx @@ -0,0 +1,141 @@ +--- +id: crawling-links +title: Crawling links +description: How to crawl a list of URLs, discover and enqueue new links on the fly, and control which links get followed. +--- + +import ApiLink from '@site/src/components/ApiLink'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; +import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; + +import MultipleUrlsBeautifulSoupExample from '!!raw-loader!roa-loader!./code_examples/crawling_links/crawl_multiple_urls_bs.py'; +import MultipleUrlsPlaywrightExample from '!!raw-loader!roa-loader!./code_examples/crawling_links/crawl_multiple_urls_pw.py'; + +import AllLinksBeautifulSoupExample from '!!raw-loader!roa-loader!./code_examples/crawling_links/crawl_all_links_on_website_bs.py'; +import AllLinksPlaywrightExample from '!!raw-loader!roa-loader!./code_examples/crawling_links/crawl_all_links_on_website_pw.py'; + +import SpecificLinksBeautifulSoupExample from '!!raw-loader!roa-loader!./code_examples/crawling_links/crawl_specific_links_on_website_bs.py'; +import SpecificLinksPlaywrightExample from '!!raw-loader!roa-loader!./code_examples/crawling_links/crawl_specific_links_on_website_pw.py'; + +import ExtractAndAddBeautifulSoupExample from '!!raw-loader!roa-loader!./code_examples/crawling_links/extract_and_add_specific_links_on_website_bs.py'; +import ExtractAndAddPlaywrightExample from '!!raw-loader!roa-loader!./code_examples/crawling_links/extract_and_add_specific_links_on_website_pw.py'; + +import StrategyAllLinksExample from '!!raw-loader!roa-loader!./code_examples/crawling_links/crawl_website_with_relative_links_all_links.py'; +import StrategySameDomainExample from '!!raw-loader!roa-loader!./code_examples/crawling_links/crawl_website_with_relative_links_same_domain.py'; +import StrategySameHostnameExample from '!!raw-loader!roa-loader!./code_examples/crawling_links/crawl_website_with_relative_links_same_hostname.py'; +import StrategySameOriginExample from '!!raw-loader!roa-loader!./code_examples/crawling_links/crawl_website_with_relative_links_same_origin.py'; + +Most crawls start from a handful of URLs and grow by following links discovered on the visited pages. This guide shows how to crawl a fixed list of URLs, how to enqueue newly discovered links with the `enqueue_links` helper, and how to control which links end up in the `RequestQueue`. + +## Crawling a list of URLs + +The simplest case is crawling a known list of URLs. Pass the list to `crawler.run` and the request handler is invoked for each of them. + + + + + {MultipleUrlsBeautifulSoupExample} + + + + + {MultipleUrlsPlaywrightExample} + + + + +## Crawling all links on a website + +To crawl a whole website, call the `enqueue_links` helper in the request handler. It finds the links on the current page and adds them to the `RequestQueue`, so the crawler systematically works through the site page by page. + +:::tip + +If no options are given, by default the method will only add links that are under the same hostname. This behavior can be controlled with the `strategy` option. For details, see [Enqueue strategies](#enqueue-strategies). + +::: + + + + + {AllLinksBeautifulSoupExample} + + + + + {AllLinksPlaywrightExample} + + + + +## Crawling specific links + +You'll often want to follow only links that match certain patterns. Pass the `include` or `exclude` parameters to `enqueue_links` - both accept lists of globs or regular expressions - and only the matching links are added to the `RequestQueue`. This keeps the crawl focused on the relevant sections of a website and avoids scraping unnecessary content. + + + + + {SpecificLinksBeautifulSoupExample} + + + + + {SpecificLinksPlaywrightExample} + + + + +### Even more control over the enqueued links + +`enqueue_links` is a convenience helper. Internally it calls `extract_links` to find the links and `add_requests` to add them to the queue. If you need custom filtering of the extracted links before enqueuing them, use `extract_links` and `add_requests` directly: + + + + + {ExtractAndAddBeautifulSoupExample} + + + + + {ExtractAndAddPlaywrightExample} + + + + +## Enqueue strategies + +`enqueue_links` automatically resolves relative links based on the page's context and decides which of them to follow according to the `strategy` option (the `EnqueueStrategy` type alias). Four strategies are available: + +- `all` - Enqueues all links found, regardless of the domain they point to. Useful when you want to follow every link, including those that navigate to external websites. +- `same-domain` - Enqueues all links found that share the same domain name, including any possible subdomains. +- `same-hostname` - Enqueues all links found for the exact same hostname. This is the **default** strategy. It restricts the crawl to links with the same hostname as the current page and excludes subdomains. +- `same-origin` - Enqueues all links found that share the same origin - the same protocol, domain, and port - ensuring a strict scope for the crawl. + +:::note + +These examples use the `BeautifulSoupCrawler`, but the same method is available for all crawlers and works exactly the same way. + +::: + + + + + {StrategyAllLinksExample} + + + + + {StrategySameDomainExample} + + + + + {StrategySameHostnameExample} + + + + + {StrategySameOriginExample} + + + diff --git a/docs/examples/fill_and_submit_web_form.mdx b/docs/04_guides/02_fill_and_submit_web_form.mdx similarity index 94% rename from docs/examples/fill_and_submit_web_form.mdx rename to docs/04_guides/02_fill_and_submit_web_form.mdx index 841a2616ee..6c8226f50b 100644 --- a/docs/examples/fill_and_submit_web_form.mdx +++ b/docs/04_guides/02_fill_and_submit_web_form.mdx @@ -1,6 +1,7 @@ --- id: fill-and-submit-web-form -title: Fill and submit web form +title: Filling and submitting web forms +description: How to inspect a web form, prepare a POST request from its fields, and submit it with an HTTP crawler. --- import ApiLink from '@site/src/components/ApiLink'; @@ -8,8 +9,8 @@ import Tabs from '@theme/Tabs'; import TabItem from '@theme/TabItem'; import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; -import RequestExample from '!!raw-loader!roa-loader!./code_examples/fill_and_submit_web_form_request.py'; -import CrawlerExample from '!!raw-loader!roa-loader!./code_examples/fill_and_submit_web_form_crawler.py'; +import RequestExample from '!!raw-loader!roa-loader!./code_examples/fill_and_submit_web_form/fill_and_submit_web_form_request.py'; +import CrawlerExample from '!!raw-loader!roa-loader!./code_examples/fill_and_submit_web_form/fill_and_submit_web_form_crawler.py'; This example demonstrates how to fill and submit a web form using the `HttpCrawler` crawler. The same approach applies to any crawler that inherits from it, such as the `BeautifulSoupCrawler` or `ParselCrawler`. diff --git a/docs/guides/crawler_login.mdx b/docs/04_guides/03_crawler_login.mdx similarity index 57% rename from docs/guides/crawler_login.mdx rename to docs/04_guides/03_crawler_login.mdx index 3dcdc7c72d..197a017f89 100644 --- a/docs/guides/crawler_login.mdx +++ b/docs/04_guides/03_crawler_login.mdx @@ -5,16 +5,19 @@ description: How to log in to websites with Crawlee. --- import ApiLink from '@site/src/components/ApiLink'; +import CodeBlock from '@theme/CodeBlock'; import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; import PlaywrightLogin from '!!raw-loader!roa-loader!./code_examples/login_crawler/playwright_login.py'; import HttpLogin from '!!raw-loader!roa-loader!./code_examples/login_crawler/http_login.py'; +import ChromeProfileExample from '!!raw-loader!./code_examples/login_crawler/using_browser_profiles_chrome.py'; +import FirefoxProfileExample from '!!raw-loader!./code_examples/login_crawler/using_browser_profiles_firefox.py'; Many websites require authentication to access their content. This guide demonstrates how to implement login functionality using both `PlaywrightCrawler` and `HttpCrawler`. ## Session management for authentication -When implementing authentication, you'll typically want to maintain the same `Session` throughout your crawl to preserve login state. This requires proper configuration of the `SessionPool`. For more details, see our [session management guide](./session-management). To work with the cookies themselves, see [Cookie management](./cookie-management). +When implementing authentication, you'll typically want to maintain the same `Session` throughout your crawl to preserve login state. This requires proper configuration of the `SessionPool`. For more details, see the [Session management](../concepts/session-management) page. To work with the cookies themselves, see [Cookie management](../concepts/cookie-management). If your use case requires multiple authenticated sessions with different credentials, you can: @@ -40,3 +43,33 @@ HTTP-based authentication often varies significantly between websites. Using bro {HttpLogin} + +## Using browser profiles + +Instead of automating the login flow, you can run `PlaywrightCrawler` with your local browser profile from [Chrome](https://www.google.com/intl/us/chrome/) or [Firefox](https://www.firefox.com/). Browser profiles carry existing login sessions, saved passwords, and other personalized browser data, which lets the crawler access content that requires authentication without logging in programmatically. + +### Chrome + +To run the crawler with your Chrome profile, you need to know the path to your profile files. You can find it by entering `chrome://version/` as a URL in your Chrome browser. If you have multiple profiles, pay attention to the profile name - with a single profile, it's always `Default`. + +:::warning Profile access limitation + +Due to [Chrome's security policies](https://developer.chrome.com/blog/remote-debugging-port), automation cannot use your main browsing profile directly. The example copies your profile to a temporary location as a workaround. + +::: + +Make sure you don't have any running Chrome browser processes before running this code: + + + {ChromeProfileExample} + + +### Firefox + +To find the path to your Firefox profile, enter `about:profiles` as a URL in your Firefox browser. Unlike Chrome, you can use your standard profile path directly without copying it first. + +Make sure you don't have any running Firefox browser processes before running this code: + + + {FirefoxProfileExample} + diff --git a/docs/guides/avoid_blocking.mdx b/docs/04_guides/04_avoid_blocking.mdx similarity index 70% rename from docs/guides/avoid_blocking.mdx rename to docs/04_guides/04_avoid_blocking.mdx index f800d11c45..2204cd2520 100644 --- a/docs/guides/avoid_blocking.mdx +++ b/docs/04_guides/04_avoid_blocking.mdx @@ -1,7 +1,7 @@ --- id: avoid-blocking title: Avoid getting blocked -description: How to avoid getting blocked when scraping +description: How to avoid getting blocked by anti-bot protections, using browser fingerprints and stealth browsers. --- import ApiLink from '@site/src/components/ApiLink'; @@ -9,12 +9,12 @@ import CodeBlock from '@theme/CodeBlock'; import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; import PlaywrightDefaultFingerprintGenerator from '!!raw-loader!roa-loader!./code_examples/avoid_blocking/playwright_with_fingerprint_generator.py'; -import PlaywrightWithCamoufox from '!!raw-loader!roa-loader!../examples/code_examples/playwright_crawler_with_camoufox.py'; +import PlaywrightWithCamoufox from '!!raw-loader!roa-loader!./code_examples/avoid_blocking/playwright_crawler_with_camoufox.py'; import PlaywrightWithCloakBrowser from '!!raw-loader!roa-loader!./code_examples/avoid_blocking/playwright_with_cloakbrowser.py'; import PlaywrightDefaultFingerprintGeneratorWithArgs from '!!raw-loader!./code_examples/avoid_blocking/default_fingerprint_generator_with_args.py'; -A scraper might get blocked for numerous reasons. Let's narrow it down to the two main ones. The first is a bad or blocked IP address. You can learn about this topic in the [proxy management guide](./proxy-management). The second reason is [browser fingerprints](https://pixelprivacy.com/resources/browser-fingerprinting/) (or signatures), which we will explore more in this guide. Check the [Apify Academy anti-scraping course](https://docs.apify.com/academy/anti-scraping) to gain a deeper theoretical understanding of blocking and learn a few tips and tricks. +A scraper might get blocked for numerous reasons. Let's narrow it down to the two main ones. The first is a bad or blocked IP address. You can learn about this topic on the [Proxy management](../concepts/proxy-management) page. The second reason is [browser fingerprints](https://pixelprivacy.com/resources/browser-fingerprinting/) (or signatures), which we will explore more in this guide. Check the [Apify Academy anti-scraping course](https://docs.apify.com/academy/anti-scraping) to gain a deeper theoretical understanding of blocking and learn a few tips and tricks. Browser fingerprint is a collection of browser attributes and significant features that can show if our browser is a bot or a real user. Moreover, most browsers have these unique features that allow the website to track the browser even within different IP addresses. This is the main reason why scrapers should change browser fingerprints while doing browser-based scraping. In return, it should significantly reduce the blocking. @@ -36,7 +36,15 @@ If you do not want to use fingerprints, then pass `fingerprint_generator=None` a ## Using Camoufox -In some cases even `PlaywrightCrawler` with fingerprints is not enough. You can try using `PlaywrightCrawler` together with [Camoufox](https://camoufox.com/). See the example integration below: +In some cases even `PlaywrightCrawler` with fingerprints is not enough. You can try using `PlaywrightCrawler` together with [Camoufox](https://camoufox.com/), a stealthy, minimalistic build of Firefox. The example below integrates Camoufox into `PlaywrightCrawler` using `BrowserPool` with a custom `PlaywrightBrowserPlugin`. + +Camoufox is an external tool and is not part of Crawlee, so [install it separately](https://pypi.org/project/camoufox/). Note that Camoufox uses a custom build of Firefox which can be hundreds of MB large. You can pre-download it with `python3 -m camoufox fetch`. Otherwise Camoufox downloads it automatically on first run. See the [Camoufox Python interface documentation](https://github.com/daijro/camoufox/tree/main/pythonlib#camoufox-python-interface) for details. + +:::tip + +You can generate a project with the Camoufox integration already set up through the Crawlee CLI. Run `crawlee create` and pick `Playwright-camoufox` when asked for the crawler type. + +::: {PlaywrightWithCamoufox} diff --git a/docs/examples/respect_robots_txt_file.mdx b/docs/04_guides/05_respect_robots_txt_file.mdx similarity index 86% rename from docs/examples/respect_robots_txt_file.mdx rename to docs/04_guides/05_respect_robots_txt_file.mdx index dc509e16b8..84f2f714d2 100644 --- a/docs/examples/respect_robots_txt_file.mdx +++ b/docs/04_guides/05_respect_robots_txt_file.mdx @@ -1,13 +1,14 @@ --- id: respect-robots-txt-file -title: Respect robots.txt file +title: Respecting robots.txt +description: How to make a crawler respect the rules websites declare for bots in their robots.txt file. --- import ApiLink from '@site/src/components/ApiLink'; import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; -import RespectRobotsTxt from '!!raw-loader!roa-loader!./code_examples/respect_robots_txt_file.py'; -import OnSkippedRequest from '!!raw-loader!roa-loader!./code_examples/respect_robots_on_skipped_request.py'; +import RespectRobotsTxt from '!!raw-loader!roa-loader!./code_examples/respect_robots_txt_file/respect_robots_txt_file.py'; +import OnSkippedRequest from '!!raw-loader!roa-loader!./code_examples/respect_robots_txt_file/respect_robots_on_skipped_request.py'; This example demonstrates how to configure your crawler to respect the rules established by websites for crawlers as described in the [robots.txt](https://www.robotstxt.org/robotstxt.html) file. diff --git a/docs/04_guides/06_stopping_and_resuming.mdx b/docs/04_guides/06_stopping_and_resuming.mdx new file mode 100644 index 0000000000..2a05f28420 --- /dev/null +++ b/docs/04_guides/06_stopping_and_resuming.mdx @@ -0,0 +1,56 @@ +--- +id: stopping-and-resuming-crawlers +title: Stopping and resuming crawlers +description: How to stop a crawler from a request handler, keep it alive while waiting for more requests, and resume an interrupted crawl. +--- + +import ApiLink from '@site/src/components/ApiLink'; +import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; + +import CrawlerStopExample from '!!raw-loader!roa-loader!./code_examples/stopping_and_resuming/beautifulsoup_crawler_stop.py'; +import CrawlerKeepAliveExample from '!!raw-loader!roa-loader!./code_examples/stopping_and_resuming/beautifulsoup_crawler_keep_alive.py'; +import ResumeCrawlExample from '!!raw-loader!roa-loader!./code_examples/stopping_and_resuming/resuming_paused_crawl.py'; + +A crawler normally runs until its request queue is empty. This guide covers the cases where you want to change that: stopping the crawl early once you've found what you're looking for, keeping the crawler alive while it waits for more requests, and resuming an interrupted crawl from where it stopped. + +All the options on this page are available to every crawler that inherits from `BasicCrawler`. The examples use `BeautifulSoupCrawler`, but they work the same with any other crawler. + +## Stopping a crawler + +Call `crawler.stop()` to stop the crawler, for example once it finds what it's looking for. The crawler won't pick up any new requests, but requests that are already being processed concurrently are finished. The optional `reason` argument is a string that appears in the logs, which can improve their readability, especially when multiple conditions can trigger the stop. + + + {CrawlerStopExample} + + +## Keeping a crawler alive + +To keep the crawler running even when there are no requests to process at the moment, pass `keep_alive=True` to the crawler's constructor (see `BasicCrawlerOptions`). The crawler then waits for more requests to be added instead of finishing. This setup is useful when another component adds requests over time, as in the [Running in web server](./running-in-web-server) guide. A crawler started with `keep_alive=True` is stopped by calling `crawler.stop()`. + + + {CrawlerKeepAliveExample} + + +## Resuming a paused crawl + +When a local crawler run is interrupted - say with `CTRL+C` - the default behavior is to start over on the next run, because Crawlee purges the storage on start. To continue from the previous state instead, disable `purge_on_start` in the `Configuration`. + +Use the following code and perform two sequential runs. During the first run, stop the crawler by pressing `CTRL+C`, and the second run will resume crawling from where it stopped. + + + {ResumeCrawlExample} + + +Perform the first run, interrupting the crawler with `CTRL+C` after 2 links have been processed. + +![Run with interruption](/img/resuming-paused-crawl/00.webp 'Run with interruption.') + +Now resume crawling after the pause to process the remaining 3 links. + +![Resuming crawling](/img/resuming-paused-crawl/01.webp 'Resuming crawling.') + +Alternatively, set the environment variable `CRAWLEE_PURGE_ON_START=0` instead of `configuration.purge_on_start = False`: + +```bash +CRAWLEE_PURGE_ON_START=0 python your_crawler.py +``` diff --git a/docs/examples/run_parallel_crawlers.mdx b/docs/04_guides/07_run_parallel_crawlers.mdx similarity index 76% rename from docs/examples/run_parallel_crawlers.mdx rename to docs/04_guides/07_run_parallel_crawlers.mdx index fba5c437b7..345eca8fd3 100644 --- a/docs/examples/run_parallel_crawlers.mdx +++ b/docs/04_guides/07_run_parallel_crawlers.mdx @@ -1,18 +1,19 @@ --- id: run-parallel-crawlers -title: Run parallel crawlers +title: Running crawlers in parallel +description: How to run two crawlers in parallel and pass links between them through request queues. --- import ApiLink from '@site/src/components/ApiLink'; import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; -import RunParallelCrawlersExample from '!!raw-loader!roa-loader!./code_examples/run_parallel_crawlers.py'; +import RunParallelCrawlersExample from '!!raw-loader!roa-loader!./code_examples/run_parallel_crawlers/run_parallel_crawlers.py'; This example demonstrates how to run two parallel crawlers where one crawler processes links discovered by another crawler. -In some situations, you may need different approaches for scraping data from a website. For example, you might use `PlaywrightCrawler` for navigating JavaScript-heavy pages and a faster, more lightweight `ParselCrawler` for processing static pages. One way to solve this is to use `AdaptivePlaywrightCrawler`, see the [Adaptive Playwright crawler example](./adaptive-playwright-crawler) to learn more. +In some situations, you may need different approaches for scraping data from a website. For example, you might use `PlaywrightCrawler` for navigating JavaScript-heavy pages and a faster, more lightweight `ParselCrawler` for processing static pages. One way to solve this is to use `AdaptivePlaywrightCrawler`, see the [Adaptive Playwright crawler](../concepts/adaptive-playwright-crawler) page to learn more. -The code below demonstrates an alternative approach using two separate crawlers. Links are passed between crawlers via `RequestQueue` aliases. The `keep_alive` option allows the Playwright crawler to run in the background and wait for incoming links without stopping when its queue is empty. You can also use different storage clients for each crawler without losing the ability to pass links between queues. Learn more about available storage clients in this [guide](/python/docs/guides/storage-clients). +The code below demonstrates an alternative approach using two separate crawlers. Links are passed between crawlers via `RequestQueue` aliases. The `keep_alive` option allows the Playwright crawler to run in the background and wait for incoming links without stopping when its queue is empty. You can also use different storage clients for each crawler without losing the ability to pass links between queues. Learn more about available storage clients on the [Storage clients](../concepts/storage-clients) page. {RunParallelCrawlersExample} diff --git a/docs/guides/running_in_web_server.mdx b/docs/04_guides/08_running_in_web_server.mdx similarity index 96% rename from docs/guides/running_in_web_server.mdx rename to docs/04_guides/08_running_in_web_server.mdx index 8e29bf4cb7..be3f1465b1 100644 --- a/docs/guides/running_in_web_server.mdx +++ b/docs/04_guides/08_running_in_web_server.mdx @@ -1,7 +1,7 @@ --- id: running-in-web-server title: Running in web server -description: Running in web server +description: How to run a crawler inside a web server and return crawl results in response to HTTP requests. --- import ApiLink from '@site/src/components/ApiLink'; diff --git a/docs/guides/trace_and_monitor_crawlers.mdx b/docs/04_guides/09_trace_and_monitor_crawlers.mdx similarity index 100% rename from docs/guides/trace_and_monitor_crawlers.mdx rename to docs/04_guides/09_trace_and_monitor_crawlers.mdx diff --git a/docs/guides/creating_web_archive.mdx b/docs/04_guides/10_creating_web_archive.mdx similarity index 98% rename from docs/guides/creating_web_archive.mdx rename to docs/04_guides/10_creating_web_archive.mdx index bafc8ec888..6040d78d80 100644 --- a/docs/guides/creating_web_archive.mdx +++ b/docs/04_guides/10_creating_web_archive.mdx @@ -1,7 +1,7 @@ --- id: creating-web-archive title: Creating web archive -description: How to create a Web ARChive (WARC) with Crawlee +description: How to create a Web ARChive (WARC) of the crawled pages with Crawlee. --- import ApiLink from '@site/src/components/ApiLink'; diff --git a/docs/guides/secure_scraping.mdx b/docs/04_guides/11_secure_scraping.mdx similarity index 95% rename from docs/guides/secure_scraping.mdx rename to docs/04_guides/11_secure_scraping.mdx index 3cfa3097d6..97b9b07279 100644 --- a/docs/guides/secure_scraping.mdx +++ b/docs/04_guides/11_secure_scraping.mdx @@ -42,7 +42,7 @@ await context.enqueue_links() await context.enqueue_links(strategy='all') ``` -The same rules apply to URLs read by a `SitemapRequestLoader` and to `Sitemap:` directives in `robots.txt`. A target can't seed your queue with off-host URLs that way either. Following `robots.txt` rules is itself opt-in: `respect_robots_txt_file` is `False` by default. The [respecting robots.txt example](../examples/respect-robots-txt-file) shows how to turn it on. +The same rules apply to URLs read by a `SitemapRequestLoader` and to `Sitemap:` directives in `robots.txt`. A target can't seed your queue with off-host URLs that way either. Following `robots.txt` rules is itself opt-in: `respect_robots_txt_file` is `False` by default. The [respecting robots.txt example](./respect-robots-txt-file) shows how to turn it on. ### Widening the scope @@ -79,7 +79,7 @@ It's the same rule in every case. Treat an extracted value as untrusted until yo ## Untrusted proxies -Crawlers route requests through [proxies](./proxy-management) to rotate IPs and avoid blocking. A proxy sees and can alter every request and response that passes through it, so a free, unknown, or compromised proxy is a man-in-the-middle. It can log the URLs and data you send, read and modify responses, and capture access credentials like login submissions, API keys, `Authorization` headers, and session cookies. An attacker who collects those can reuse them to take over the accounts behind them. +Crawlers route requests through [proxies](../concepts/proxy-management) to rotate IPs and avoid blocking. A proxy sees and can alter every request and response that passes through it, so a free, unknown, or compromised proxy is a man-in-the-middle. It can log the URLs and data you send, read and modify responses, and capture access credentials like login submissions, API keys, `Authorization` headers, and session cookies. An attacker who collects those can reuse them to take over the accounts behind them. The built-in HTTP clients verify TLS certificates by default. Keeping that on and keeping traffic on HTTPS stops a proxy from reading or rewriting payloads, though it still sees the destination host through SNI and `CONNECT`. Prefer proxy providers you trust. Don't send credentials or secrets through a proxy you don't control. Don't disable certificate verification just to make an untrusted proxy work. diff --git a/docs/guides/extending_crawlee.mdx b/docs/04_guides/12_extending_crawlee.mdx similarity index 93% rename from docs/guides/extending_crawlee.mdx rename to docs/04_guides/12_extending_crawlee.mdx index 6e4a3d3977..07c4b7144f 100644 --- a/docs/guides/extending_crawlee.mdx +++ b/docs/04_guides/12_extending_crawlee.mdx @@ -124,7 +124,7 @@ The context exposes the parsed data to handlers, and the crawler ties the parser Browser crawlers use the same orchestration with a browser-backed context. Extend `PlaywrightCrawler` when an integration needs crawler-level browser behavior or a different handler context. `StagehandCrawler` is an example. It extends `PlaywrightCrawler` with a Stagehand-specific context and browser behavior. If only browser launch or lifecycle differs, a browser plugin is the narrower extension point. -See the [HTTP crawlers guide](./http-crawlers) for a worked example built on `selectolax`, and the [Stagehand crawler guide](./stagehand-crawler) for `StagehandCrawler` itself. The [Architecture overview](./architecture-overview) explains how HTTP and browser crawlers relate to the other components. +See the [HTTP crawlers](../concepts/http-crawlers) page for a worked example built on `selectolax`, and the [Stagehand crawler guide](./stagehand-crawler) for `StagehandCrawler` itself. The [Architecture overview](../concepts/architecture-overview) explains how HTTP and browser crawlers relate to the other components. ### HTTP clients @@ -139,7 +139,7 @@ The contract is `HttpClient`: Crawlee ships `ImpitHttpClient`, `HttpxHttpClient`, and `CurlImpersonateHttpClient`. -See the [HTTP clients guide](./http-clients) for the libraries these clients wrap and their installation requirements. +See the [HTTP clients](../concepts/http-clients) page for the libraries these clients wrap and their installation requirements. ### Storage clients @@ -153,7 +153,7 @@ The `StorageClient` contract defines A custom backend implements all four classes. -See the [Storage clients guide](./storage-clients) for the built-in implementations and a custom client example. +See the [Storage clients](../concepts/storage-clients) page for the built-in implementations and a custom client example. ### Browser plugins @@ -170,7 +170,7 @@ Implement this base contract directly when the launch and lifecycle are too spec Most integrations should start with `PlaywrightBrowserPlugin`. Configure it when its launch and context options cover the required browser. Extend it when you need a custom Playwright-compatible launch path while preserving its standard lifecycle and context handling. -See the [Playwright crawler guide](./playwright-crawler) for the responsibilities a subclass has to preserve, and the [Camoufox example](../examples/playwright-crawler-with-camoufox) for a complete integration. +See the [Playwright crawler](../concepts/playwright-crawler) page for the responsibilities a subclass has to preserve, and the [Camoufox example](./avoid-blocking#using-camoufox) for a complete integration. ## Choosing an extension point diff --git a/docs/guides/pydantic_ai_crawler.mdx b/docs/04_guides/13_pydantic_ai_crawler.mdx similarity index 100% rename from docs/guides/pydantic_ai_crawler.mdx rename to docs/04_guides/13_pydantic_ai_crawler.mdx diff --git a/docs/guides/stagehand_crawler.mdx b/docs/04_guides/14_stagehand_crawler.mdx similarity index 97% rename from docs/guides/stagehand_crawler.mdx rename to docs/04_guides/14_stagehand_crawler.mdx index adbd31b3e9..7bc7cab98b 100644 --- a/docs/guides/stagehand_crawler.mdx +++ b/docs/04_guides/14_stagehand_crawler.mdx @@ -29,7 +29,7 @@ Use `StagehandCrawler` when: - **Interactions are complex** - multi-step forms, dynamic menus, or context-dependent flows that are hard to script. - **Rapid prototyping** - you want to build a scraper quickly without spending time reverse-engineering the page structure. -For straightforward scraping tasks where the page structure is stable and well-known, `PlaywrightCrawler` is more efficient, read more in that [guide](./playwright-crawler). +For straightforward scraping tasks where the page structure is stable and well-known, `PlaywrightCrawler` is more efficient, read more on the [Playwright crawler](../concepts/playwright-crawler) page. ## Installation diff --git a/docs/guides/scrapy_migration.mdx b/docs/04_guides/15_scrapy_migration.mdx similarity index 92% rename from docs/guides/scrapy_migration.mdx rename to docs/04_guides/15_scrapy_migration.mdx index 5acbfa3824..7c1253daae 100644 --- a/docs/guides/scrapy_migration.mdx +++ b/docs/04_guides/15_scrapy_migration.mdx @@ -37,7 +37,7 @@ import PlaywrightExample from '!!raw-loader!roa-loader!./code_examples/scrapy_mi [Scrapy](https://scrapy.org/) and Crawlee solve the same problem, so most of what you know carries over. The concepts line up almost one to one, and the selector code often moves across without changes. Scrapy parses HTML with [Parsel](https://parsel.readthedocs.io/), and so does Crawlee's `ParselCrawler`. Your `response.css()` and `response.xpath()` queries keep working on `context.selector`. -This guide maps Scrapy concepts to their Crawlee equivalents and rewrites a small spider step by step. The examples use `ParselCrawler` because it's the closest match for a Scrapy spider. The other HTTP option is `BeautifulSoupCrawler`, and the choice between them is covered in the [HTTP crawlers guide](./http-crawlers). For pages that need a browser, Crawlee has `PlaywrightCrawler`, covered in the [Playwright crawler guide](./playwright-crawler). +This guide maps Scrapy concepts to their Crawlee equivalents and rewrites a small spider step by step. The examples use `ParselCrawler` because it's the closest match for a Scrapy spider. The other HTTP option is `BeautifulSoupCrawler`, and the choice between them is covered on the [HTTP crawlers](../concepts/http-crawlers) page. For pages that need a browser, Crawlee has `PlaywrightCrawler`, covered on the [Playwright crawler](../concepts/playwright-crawler) page. The Crawlee examples cap each run with `max_requests_per_crawl` to keep the demos short. The Scrapy equivalent is the `CLOSESPIDER_PAGECOUNT` setting from the built-in `CloseSpider` extension. @@ -123,7 +123,7 @@ Note that: ## Routing with labels -Scrapy routes pages by passing a `callback` to each request. A listing page hands its links to `parse_author`. In Crawlee you attach a `label` to the enqueued requests, then register a handler for that label. Each request goes to the matching handler, and unlabeled requests fall through to the default one. For details, see the [Request router guide](./request-router). +Scrapy routes pages by passing a `callback` to each request. A listing page hands its links to `parse_author`. In Crawlee you attach a `label` to the enqueued requests, then register a handler for that label. Each request goes to the matching handler, and unlabeled requests fall through to the default one. For details, see the [Request router](../concepts/request-router) page. @@ -191,11 +191,11 @@ Scrapy writes results through the `FEEDS` setting or the `-O` command-line flag. -The default (unnamed) `Dataset` stays on disk after the run finishes, but the next run clears it before starting, since `Configuration.purge_on_start` defaults to `True`. To keep data across runs, use a named dataset or turn that option off. For details, see the [Storages guide](./storages). To continue an interrupted crawl from where it stopped, see the [Resuming a paused crawl](../examples/resuming-paused-crawl) example. +The default (unnamed) `Dataset` stays on disk after the run finishes, but the next run clears it before starting, since `Configuration.purge_on_start` defaults to `True`. To keep data across runs, use a named dataset or turn that option off. For details, see the [Storages](../concepts/storages) page. To continue an interrupted crawl from where it stopped, see the [Resuming a paused crawl](./stopping-and-resuming-crawlers#resuming-a-paused-crawl) example. ## Concurrency and throttling -Both Scrapy and Crawlee have settings to influence the scraping throughput, but they differ in their approach. Crawlee reads `ConcurrencySettings` and scales concurrency automatically based on system resources, as described in the [Scaling crawlers guide](./scaling-crawlers). Use `max_tasks_per_minute` to cap the request rate. It's close to `DOWNLOAD_DELAY`, but it limits the whole pool rather than adding a fixed delay per domain. +Both Scrapy and Crawlee have settings to influence the scraping throughput, but they differ in their approach. Crawlee reads `ConcurrencySettings` and scales concurrency automatically based on system resources, as described on the [Scaling crawlers](../concepts/scaling-crawlers) page. Use `max_tasks_per_minute` to cap the request rate. It's close to `DOWNLOAD_DELAY`, but it limits the whole pool rather than adding a fixed delay per domain. @@ -210,7 +210,7 @@ Both Scrapy and Crawlee have settings to influence the scraping throughput, but -The closest built-in analog to the `AutoThrottle` extension is the `ThrottlingRequestManager`, covered in the [Request throttling guide](./request-throttling). It differs in what drives the backoff. Scrapy's `AutoThrottle` adapts to measured latency, while Crawlee's manager reacts to explicit signals: HTTP 429 replies and `robots.txt` crawl-delay directives. It wraps a `RequestQueue` and applies the backoff per domain. You list the domains to watch, and the crawler reports the signals to the manager on its own. +The closest built-in analog to the `AutoThrottle` extension is the `ThrottlingRequestManager`, covered on the [Request throttling](../concepts/request-throttling) page. It differs in what drives the backoff. Scrapy's `AutoThrottle` adapts to measured latency, while Crawlee's manager reacts to explicit signals: HTTP 429 replies and `robots.txt` crawl-delay directives. It wraps a `RequestQueue` and applies the backoff per domain. You list the domains to watch, and the crawler reports the signals to the manager on its own. Crawl-delay works only when the crawler fetches `robots.txt`. Projects generated by `scrapy startproject` obey `robots.txt` out of the box, but the default for Crawlee is to ignore it. Passing `respect_robots_txt_file=True` tells Crawlee to enforce the `Disallow` rules. Together with `ThrottlingRequestManager` it also reads crawl-delay, otherwise it logs a warning. @@ -229,7 +229,7 @@ Crawl-delay works only when the crawler fetches `robots.txt`. Projects generated ## Proxies -Scrapy rotates proxies through a middleware or a third-party package. Crawlee takes a `ProxyConfiguration` and rotates the URLs for you. For tiered proxies, session-bound proxies, and integration with the Apify Proxy, see the [Proxy management guide](./proxy-management). +Scrapy rotates proxies through a middleware or a third-party package. Crawlee takes a `ProxyConfiguration` and rotates the URLs for you. For tiered proxies, session-bound proxies, and integration with the Apify Proxy, see the [Proxy management](../concepts/proxy-management) page. @@ -246,7 +246,7 @@ Scrapy rotates proxies through a middleware or a third-party package. Crawlee ta ## Error handling -Scrapy retries failed requests with `RetryMiddleware` and reports terminal failures through `errback`. Crawlee retries with `max_request_retries`, runs an `error_handler` between attempts, and calls a `failed_request_handler` once the retries run out. For details, see the [Error handling guide](./error-handling). +Scrapy retries failed requests with `RetryMiddleware` and reports terminal failures through `errback`. Crawlee retries with `max_request_retries`, runs an `error_handler` between attempts, and calls a `failed_request_handler` once the retries run out. For details, see the [Error handling](../concepts/error-handling) page. @@ -280,7 +280,7 @@ Scrapy submits forms with `FormRequest`, which encodes `formdata` as `form-urlen ## JavaScript rendering -For pages that need a browser, Scrapy users add the `scrapy-playwright` package, which requires extra settings to register its download handlers and the asyncio reactor. Crawlee has browser support built in through the `PlaywrightCrawler`. The handler receives a [Playwright `Page`](https://playwright.dev/python/docs/api/class-page) on `context.page`, so you query the rendered DOM instead of a static response. For details, see the [Playwright crawler guide](./playwright-crawler). +For pages that need a browser, Scrapy users add the `scrapy-playwright` package, which requires extra settings to register its download handlers and the asyncio reactor. Crawlee has browser support built in through the `PlaywrightCrawler`. The handler receives a [Playwright `Page`](https://playwright.dev/python/docs/api/class-page) on `context.page`, so you query the rendered DOM instead of a static response. For details, see the [Playwright crawler](../concepts/playwright-crawler) page. @@ -298,7 +298,7 @@ For pages that need a browser, Scrapy users add the `scrapy-playwright` package, -`PlaywrightCrawler` renders every request. In Scrapy you flag individual requests with `meta={'playwright': True}` and leave the rest on the plain downloader. For that mix in Crawlee, use `AdaptivePlaywrightCrawler`, which renders in a browser only when the page needs it. For details, see the [Adaptive Playwright crawler guide](./adaptive-playwright-crawler). +`PlaywrightCrawler` renders every request. In Scrapy you flag individual requests with `meta={'playwright': True}` and leave the rest on the plain downloader. For that mix in Crawlee, use `AdaptivePlaywrightCrawler`, which renders in a browser only when the page needs it. For details, see the [Adaptive Playwright crawler](../concepts/adaptive-playwright-crawler) page. ## Conclusion diff --git a/docs/guides/code_examples/avoid_blocking/default_fingerprint_generator_with_args.py b/docs/04_guides/code_examples/avoid_blocking/default_fingerprint_generator_with_args.py similarity index 100% rename from docs/guides/code_examples/avoid_blocking/default_fingerprint_generator_with_args.py rename to docs/04_guides/code_examples/avoid_blocking/default_fingerprint_generator_with_args.py diff --git a/docs/examples/code_examples/playwright_crawler_with_camoufox.py b/docs/04_guides/code_examples/avoid_blocking/playwright_crawler_with_camoufox.py similarity index 100% rename from docs/examples/code_examples/playwright_crawler_with_camoufox.py rename to docs/04_guides/code_examples/avoid_blocking/playwright_crawler_with_camoufox.py diff --git a/docs/guides/code_examples/avoid_blocking/playwright_with_cloakbrowser.py b/docs/04_guides/code_examples/avoid_blocking/playwright_with_cloakbrowser.py similarity index 100% rename from docs/guides/code_examples/avoid_blocking/playwright_with_cloakbrowser.py rename to docs/04_guides/code_examples/avoid_blocking/playwright_with_cloakbrowser.py diff --git a/docs/guides/code_examples/avoid_blocking/playwright_with_fingerprint_generator.py b/docs/04_guides/code_examples/avoid_blocking/playwright_with_fingerprint_generator.py similarity index 100% rename from docs/guides/code_examples/avoid_blocking/playwright_with_fingerprint_generator.py rename to docs/04_guides/code_examples/avoid_blocking/playwright_with_fingerprint_generator.py diff --git a/docs/examples/code_examples/crawl_all_links_on_website_bs.py b/docs/04_guides/code_examples/crawling_links/crawl_all_links_on_website_bs.py similarity index 100% rename from docs/examples/code_examples/crawl_all_links_on_website_bs.py rename to docs/04_guides/code_examples/crawling_links/crawl_all_links_on_website_bs.py diff --git a/docs/examples/code_examples/crawl_all_links_on_website_pw.py b/docs/04_guides/code_examples/crawling_links/crawl_all_links_on_website_pw.py similarity index 100% rename from docs/examples/code_examples/crawl_all_links_on_website_pw.py rename to docs/04_guides/code_examples/crawling_links/crawl_all_links_on_website_pw.py diff --git a/docs/examples/code_examples/crawl_multiple_urls_bs.py b/docs/04_guides/code_examples/crawling_links/crawl_multiple_urls_bs.py similarity index 100% rename from docs/examples/code_examples/crawl_multiple_urls_bs.py rename to docs/04_guides/code_examples/crawling_links/crawl_multiple_urls_bs.py diff --git a/docs/examples/code_examples/crawl_multiple_urls_pw.py b/docs/04_guides/code_examples/crawling_links/crawl_multiple_urls_pw.py similarity index 100% rename from docs/examples/code_examples/crawl_multiple_urls_pw.py rename to docs/04_guides/code_examples/crawling_links/crawl_multiple_urls_pw.py diff --git a/docs/examples/code_examples/crawl_specific_links_on_website_bs.py b/docs/04_guides/code_examples/crawling_links/crawl_specific_links_on_website_bs.py similarity index 100% rename from docs/examples/code_examples/crawl_specific_links_on_website_bs.py rename to docs/04_guides/code_examples/crawling_links/crawl_specific_links_on_website_bs.py diff --git a/docs/examples/code_examples/crawl_specific_links_on_website_pw.py b/docs/04_guides/code_examples/crawling_links/crawl_specific_links_on_website_pw.py similarity index 100% rename from docs/examples/code_examples/crawl_specific_links_on_website_pw.py rename to docs/04_guides/code_examples/crawling_links/crawl_specific_links_on_website_pw.py diff --git a/docs/examples/code_examples/crawl_website_with_relative_links_all_links.py b/docs/04_guides/code_examples/crawling_links/crawl_website_with_relative_links_all_links.py similarity index 100% rename from docs/examples/code_examples/crawl_website_with_relative_links_all_links.py rename to docs/04_guides/code_examples/crawling_links/crawl_website_with_relative_links_all_links.py diff --git a/docs/examples/code_examples/crawl_website_with_relative_links_same_domain.py b/docs/04_guides/code_examples/crawling_links/crawl_website_with_relative_links_same_domain.py similarity index 100% rename from docs/examples/code_examples/crawl_website_with_relative_links_same_domain.py rename to docs/04_guides/code_examples/crawling_links/crawl_website_with_relative_links_same_domain.py diff --git a/docs/examples/code_examples/crawl_website_with_relative_links_same_hostname.py b/docs/04_guides/code_examples/crawling_links/crawl_website_with_relative_links_same_hostname.py similarity index 100% rename from docs/examples/code_examples/crawl_website_with_relative_links_same_hostname.py rename to docs/04_guides/code_examples/crawling_links/crawl_website_with_relative_links_same_hostname.py diff --git a/docs/examples/code_examples/crawl_website_with_relative_links_same_origin.py b/docs/04_guides/code_examples/crawling_links/crawl_website_with_relative_links_same_origin.py similarity index 100% rename from docs/examples/code_examples/crawl_website_with_relative_links_same_origin.py rename to docs/04_guides/code_examples/crawling_links/crawl_website_with_relative_links_same_origin.py diff --git a/docs/examples/code_examples/extract_and_add_specific_links_on_website_bs.py b/docs/04_guides/code_examples/crawling_links/extract_and_add_specific_links_on_website_bs.py similarity index 100% rename from docs/examples/code_examples/extract_and_add_specific_links_on_website_bs.py rename to docs/04_guides/code_examples/crawling_links/extract_and_add_specific_links_on_website_bs.py diff --git a/docs/examples/code_examples/extract_and_add_specific_links_on_website_pw.py b/docs/04_guides/code_examples/crawling_links/extract_and_add_specific_links_on_website_pw.py similarity index 100% rename from docs/examples/code_examples/extract_and_add_specific_links_on_website_pw.py rename to docs/04_guides/code_examples/crawling_links/extract_and_add_specific_links_on_website_pw.py diff --git a/docs/guides/code_examples/creating_web_archive/manual_archiving_parsel_crawler.py b/docs/04_guides/code_examples/creating_web_archive/manual_archiving_parsel_crawler.py similarity index 100% rename from docs/guides/code_examples/creating_web_archive/manual_archiving_parsel_crawler.py rename to docs/04_guides/code_examples/creating_web_archive/manual_archiving_parsel_crawler.py diff --git a/docs/guides/code_examples/creating_web_archive/manual_archiving_playwright_crawler.py b/docs/04_guides/code_examples/creating_web_archive/manual_archiving_playwright_crawler.py similarity index 100% rename from docs/guides/code_examples/creating_web_archive/manual_archiving_playwright_crawler.py rename to docs/04_guides/code_examples/creating_web_archive/manual_archiving_playwright_crawler.py diff --git a/docs/guides/code_examples/creating_web_archive/simple_pw_through_proxy_pywb_server.py b/docs/04_guides/code_examples/creating_web_archive/simple_pw_through_proxy_pywb_server.py similarity index 100% rename from docs/guides/code_examples/creating_web_archive/simple_pw_through_proxy_pywb_server.py rename to docs/04_guides/code_examples/creating_web_archive/simple_pw_through_proxy_pywb_server.py diff --git a/docs/examples/code_examples/fill_and_submit_web_form_crawler.py b/docs/04_guides/code_examples/fill_and_submit_web_form/fill_and_submit_web_form_crawler.py similarity index 100% rename from docs/examples/code_examples/fill_and_submit_web_form_crawler.py rename to docs/04_guides/code_examples/fill_and_submit_web_form/fill_and_submit_web_form_crawler.py diff --git a/docs/examples/code_examples/fill_and_submit_web_form_request.py b/docs/04_guides/code_examples/fill_and_submit_web_form/fill_and_submit_web_form_request.py similarity index 100% rename from docs/examples/code_examples/fill_and_submit_web_form_request.py rename to docs/04_guides/code_examples/fill_and_submit_web_form/fill_and_submit_web_form_request.py diff --git a/docs/guides/code_examples/login_crawler/http_login.py b/docs/04_guides/code_examples/login_crawler/http_login.py similarity index 100% rename from docs/guides/code_examples/login_crawler/http_login.py rename to docs/04_guides/code_examples/login_crawler/http_login.py diff --git a/docs/guides/code_examples/login_crawler/playwright_login.py b/docs/04_guides/code_examples/login_crawler/playwright_login.py similarity index 100% rename from docs/guides/code_examples/login_crawler/playwright_login.py rename to docs/04_guides/code_examples/login_crawler/playwright_login.py diff --git a/docs/examples/code_examples/using_browser_profiles_chrome.py b/docs/04_guides/code_examples/login_crawler/using_browser_profiles_chrome.py similarity index 100% rename from docs/examples/code_examples/using_browser_profiles_chrome.py rename to docs/04_guides/code_examples/login_crawler/using_browser_profiles_chrome.py diff --git a/docs/examples/code_examples/using_browser_profiles_firefox.py b/docs/04_guides/code_examples/login_crawler/using_browser_profiles_firefox.py similarity index 100% rename from docs/examples/code_examples/using_browser_profiles_firefox.py rename to docs/04_guides/code_examples/login_crawler/using_browser_profiles_firefox.py diff --git a/docs/guides/code_examples/pydantic_ai_crawler/additional_instructions_example.py b/docs/04_guides/code_examples/pydantic_ai_crawler/additional_instructions_example.py similarity index 100% rename from docs/guides/code_examples/pydantic_ai_crawler/additional_instructions_example.py rename to docs/04_guides/code_examples/pydantic_ai_crawler/additional_instructions_example.py diff --git a/docs/guides/code_examples/pydantic_ai_crawler/basic_example.py b/docs/04_guides/code_examples/pydantic_ai_crawler/basic_example.py similarity index 100% rename from docs/guides/code_examples/pydantic_ai_crawler/basic_example.py rename to docs/04_guides/code_examples/pydantic_ai_crawler/basic_example.py diff --git a/docs/guides/code_examples/pydantic_ai_crawler/custom_distiller_example.py b/docs/04_guides/code_examples/pydantic_ai_crawler/custom_distiller_example.py similarity index 100% rename from docs/guides/code_examples/pydantic_ai_crawler/custom_distiller_example.py rename to docs/04_guides/code_examples/pydantic_ai_crawler/custom_distiller_example.py diff --git a/docs/guides/code_examples/pydantic_ai_crawler/debugging_example.py b/docs/04_guides/code_examples/pydantic_ai_crawler/debugging_example.py similarity index 100% rename from docs/guides/code_examples/pydantic_ai_crawler/debugging_example.py rename to docs/04_guides/code_examples/pydantic_ai_crawler/debugging_example.py diff --git a/docs/guides/code_examples/pydantic_ai_crawler/selector_extractor_example.py b/docs/04_guides/code_examples/pydantic_ai_crawler/selector_extractor_example.py similarity index 100% rename from docs/guides/code_examples/pydantic_ai_crawler/selector_extractor_example.py rename to docs/04_guides/code_examples/pydantic_ai_crawler/selector_extractor_example.py diff --git a/docs/guides/code_examples/pydantic_ai_crawler/usage_limit_example.py b/docs/04_guides/code_examples/pydantic_ai_crawler/usage_limit_example.py similarity index 100% rename from docs/guides/code_examples/pydantic_ai_crawler/usage_limit_example.py rename to docs/04_guides/code_examples/pydantic_ai_crawler/usage_limit_example.py diff --git a/docs/examples/code_examples/respect_robots_on_skipped_request.py b/docs/04_guides/code_examples/respect_robots_txt_file/respect_robots_on_skipped_request.py similarity index 100% rename from docs/examples/code_examples/respect_robots_on_skipped_request.py rename to docs/04_guides/code_examples/respect_robots_txt_file/respect_robots_on_skipped_request.py diff --git a/docs/examples/code_examples/respect_robots_txt_file.py b/docs/04_guides/code_examples/respect_robots_txt_file/respect_robots_txt_file.py similarity index 100% rename from docs/examples/code_examples/respect_robots_txt_file.py rename to docs/04_guides/code_examples/respect_robots_txt_file/respect_robots_txt_file.py diff --git a/docs/examples/code_examples/run_parallel_crawlers.py b/docs/04_guides/code_examples/run_parallel_crawlers/run_parallel_crawlers.py similarity index 100% rename from docs/examples/code_examples/run_parallel_crawlers.py rename to docs/04_guides/code_examples/run_parallel_crawlers/run_parallel_crawlers.py diff --git a/docs/introduction/code_examples/__init__.py b/docs/04_guides/code_examples/running_in_web_server/__init__.py similarity index 100% rename from docs/introduction/code_examples/__init__.py rename to docs/04_guides/code_examples/running_in_web_server/__init__.py diff --git a/docs/guides/code_examples/running_in_web_server/crawler.py b/docs/04_guides/code_examples/running_in_web_server/crawler.py similarity index 100% rename from docs/guides/code_examples/running_in_web_server/crawler.py rename to docs/04_guides/code_examples/running_in_web_server/crawler.py diff --git a/docs/guides/code_examples/running_in_web_server/server.py b/docs/04_guides/code_examples/running_in_web_server/server.py similarity index 100% rename from docs/guides/code_examples/running_in_web_server/server.py rename to docs/04_guides/code_examples/running_in_web_server/server.py diff --git a/docs/guides/code_examples/scrapy_migration/crawlee_concurrency.py b/docs/04_guides/code_examples/scrapy_migration/crawlee_concurrency.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/crawlee_concurrency.py rename to docs/04_guides/code_examples/scrapy_migration/crawlee_concurrency.py diff --git a/docs/guides/code_examples/scrapy_migration/crawlee_crawlspider.py b/docs/04_guides/code_examples/scrapy_migration/crawlee_crawlspider.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/crawlee_crawlspider.py rename to docs/04_guides/code_examples/scrapy_migration/crawlee_crawlspider.py diff --git a/docs/guides/code_examples/scrapy_migration/crawlee_error_handling.py b/docs/04_guides/code_examples/scrapy_migration/crawlee_error_handling.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/crawlee_error_handling.py rename to docs/04_guides/code_examples/scrapy_migration/crawlee_error_handling.py diff --git a/docs/guides/code_examples/scrapy_migration/crawlee_export.py b/docs/04_guides/code_examples/scrapy_migration/crawlee_export.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/crawlee_export.py rename to docs/04_guides/code_examples/scrapy_migration/crawlee_export.py diff --git a/docs/guides/code_examples/scrapy_migration/crawlee_labels.py b/docs/04_guides/code_examples/scrapy_migration/crawlee_labels.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/crawlee_labels.py rename to docs/04_guides/code_examples/scrapy_migration/crawlee_labels.py diff --git a/docs/guides/code_examples/scrapy_migration/crawlee_playwright.py b/docs/04_guides/code_examples/scrapy_migration/crawlee_playwright.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/crawlee_playwright.py rename to docs/04_guides/code_examples/scrapy_migration/crawlee_playwright.py diff --git a/docs/guides/code_examples/scrapy_migration/crawlee_post.py b/docs/04_guides/code_examples/scrapy_migration/crawlee_post.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/crawlee_post.py rename to docs/04_guides/code_examples/scrapy_migration/crawlee_post.py diff --git a/docs/guides/code_examples/scrapy_migration/crawlee_proxy.py b/docs/04_guides/code_examples/scrapy_migration/crawlee_proxy.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/crawlee_proxy.py rename to docs/04_guides/code_examples/scrapy_migration/crawlee_proxy.py diff --git a/docs/guides/code_examples/scrapy_migration/crawlee_quotes.py b/docs/04_guides/code_examples/scrapy_migration/crawlee_quotes.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/crawlee_quotes.py rename to docs/04_guides/code_examples/scrapy_migration/crawlee_quotes.py diff --git a/docs/guides/code_examples/scrapy_migration/crawlee_throttling.py b/docs/04_guides/code_examples/scrapy_migration/crawlee_throttling.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/crawlee_throttling.py rename to docs/04_guides/code_examples/scrapy_migration/crawlee_throttling.py diff --git a/docs/guides/code_examples/scrapy_migration/crawlee_user_data.py b/docs/04_guides/code_examples/scrapy_migration/crawlee_user_data.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/crawlee_user_data.py rename to docs/04_guides/code_examples/scrapy_migration/crawlee_user_data.py diff --git a/docs/guides/code_examples/scrapy_migration/scrapy_authors.py b/docs/04_guides/code_examples/scrapy_migration/scrapy_authors.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/scrapy_authors.py rename to docs/04_guides/code_examples/scrapy_migration/scrapy_authors.py diff --git a/docs/guides/code_examples/scrapy_migration/scrapy_cb_kwargs.py b/docs/04_guides/code_examples/scrapy_migration/scrapy_cb_kwargs.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/scrapy_cb_kwargs.py rename to docs/04_guides/code_examples/scrapy_migration/scrapy_cb_kwargs.py diff --git a/docs/guides/code_examples/scrapy_migration/scrapy_concurrency.py b/docs/04_guides/code_examples/scrapy_migration/scrapy_concurrency.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/scrapy_concurrency.py rename to docs/04_guides/code_examples/scrapy_migration/scrapy_concurrency.py diff --git a/docs/guides/code_examples/scrapy_migration/scrapy_crawlspider.py b/docs/04_guides/code_examples/scrapy_migration/scrapy_crawlspider.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/scrapy_crawlspider.py rename to docs/04_guides/code_examples/scrapy_migration/scrapy_crawlspider.py diff --git a/docs/guides/code_examples/scrapy_migration/scrapy_errback.py b/docs/04_guides/code_examples/scrapy_migration/scrapy_errback.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/scrapy_errback.py rename to docs/04_guides/code_examples/scrapy_migration/scrapy_errback.py diff --git a/docs/guides/code_examples/scrapy_migration/scrapy_export.py b/docs/04_guides/code_examples/scrapy_migration/scrapy_export.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/scrapy_export.py rename to docs/04_guides/code_examples/scrapy_migration/scrapy_export.py diff --git a/docs/guides/code_examples/scrapy_migration/scrapy_formrequest.py b/docs/04_guides/code_examples/scrapy_migration/scrapy_formrequest.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/scrapy_formrequest.py rename to docs/04_guides/code_examples/scrapy_migration/scrapy_formrequest.py diff --git a/docs/guides/code_examples/scrapy_migration/scrapy_playwright.py b/docs/04_guides/code_examples/scrapy_migration/scrapy_playwright.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/scrapy_playwright.py rename to docs/04_guides/code_examples/scrapy_migration/scrapy_playwright.py diff --git a/docs/guides/code_examples/scrapy_migration/scrapy_playwright_settings.py b/docs/04_guides/code_examples/scrapy_migration/scrapy_playwright_settings.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/scrapy_playwright_settings.py rename to docs/04_guides/code_examples/scrapy_migration/scrapy_playwright_settings.py diff --git a/docs/guides/code_examples/scrapy_migration/scrapy_proxy.py b/docs/04_guides/code_examples/scrapy_migration/scrapy_proxy.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/scrapy_proxy.py rename to docs/04_guides/code_examples/scrapy_migration/scrapy_proxy.py diff --git a/docs/guides/code_examples/scrapy_migration/scrapy_quotes.py b/docs/04_guides/code_examples/scrapy_migration/scrapy_quotes.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/scrapy_quotes.py rename to docs/04_guides/code_examples/scrapy_migration/scrapy_quotes.py diff --git a/docs/guides/code_examples/scrapy_migration/scrapy_throttling.py b/docs/04_guides/code_examples/scrapy_migration/scrapy_throttling.py similarity index 100% rename from docs/guides/code_examples/scrapy_migration/scrapy_throttling.py rename to docs/04_guides/code_examples/scrapy_migration/scrapy_throttling.py diff --git a/docs/guides/code_examples/stagehand_crawler/basic_example.py b/docs/04_guides/code_examples/stagehand_crawler/basic_example.py similarity index 100% rename from docs/guides/code_examples/stagehand_crawler/basic_example.py rename to docs/04_guides/code_examples/stagehand_crawler/basic_example.py diff --git a/docs/guides/code_examples/stagehand_crawler/browserbase_example.py b/docs/04_guides/code_examples/stagehand_crawler/browserbase_example.py similarity index 100% rename from docs/guides/code_examples/stagehand_crawler/browserbase_example.py rename to docs/04_guides/code_examples/stagehand_crawler/browserbase_example.py diff --git a/docs/examples/code_examples/beautifulsoup_crawler_keep_alive.py b/docs/04_guides/code_examples/stopping_and_resuming/beautifulsoup_crawler_keep_alive.py similarity index 100% rename from docs/examples/code_examples/beautifulsoup_crawler_keep_alive.py rename to docs/04_guides/code_examples/stopping_and_resuming/beautifulsoup_crawler_keep_alive.py diff --git a/docs/examples/code_examples/beautifulsoup_crawler_stop.py b/docs/04_guides/code_examples/stopping_and_resuming/beautifulsoup_crawler_stop.py similarity index 100% rename from docs/examples/code_examples/beautifulsoup_crawler_stop.py rename to docs/04_guides/code_examples/stopping_and_resuming/beautifulsoup_crawler_stop.py diff --git a/docs/examples/code_examples/resuming_paused_crawl.py b/docs/04_guides/code_examples/stopping_and_resuming/resuming_paused_crawl.py similarity index 100% rename from docs/examples/code_examples/resuming_paused_crawl.py rename to docs/04_guides/code_examples/stopping_and_resuming/resuming_paused_crawl.py diff --git a/docs/guides/code_examples/trace_and_monitor_crawlers/instrument_crawler.py b/docs/04_guides/code_examples/trace_and_monitor_crawlers/instrument_crawler.py similarity index 100% rename from docs/guides/code_examples/trace_and_monitor_crawlers/instrument_crawler.py rename to docs/04_guides/code_examples/trace_and_monitor_crawlers/instrument_crawler.py diff --git a/docs/deployment/apify_platform.mdx b/docs/05_deployment/apify_platform.mdx similarity index 97% rename from docs/deployment/apify_platform.mdx rename to docs/05_deployment/apify_platform.mdx index 426bcc3d2a..88db05ca7d 100644 --- a/docs/deployment/apify_platform.mdx +++ b/docs/05_deployment/apify_platform.mdx @@ -13,7 +13,7 @@ import CrawlerAsActorExample from '!!raw-loader!./code_examples/apify/crawler_as import ProxyExample from '!!raw-loader!./code_examples/apify/proxy_example.py'; import ProxyAdvancedExample from '!!raw-loader!./code_examples/apify/proxy_advanced_example.py'; -Apify is a [platform](https://apify.com) built to serve large-scale and high-performance web scraping and automation needs. It provides easy access to [compute instances (Actors)](#what-is-an-actor), convenient request and result storages, [proxies](../guides/proxy-management), scheduling, webhooks, and [more in the Apify documentation](https://docs.apify.com/), accessible through a [web interface](https://console.apify.com) or an [API](https://docs.apify.com/api). +Apify is a [platform](https://apify.com) built to serve large-scale and high-performance web scraping and automation needs. It provides easy access to [compute instances (Actors)](#what-is-an-actor), convenient request and result storages, [proxies](../concepts/proxy-management), scheduling, webhooks, and [more in the Apify documentation](https://docs.apify.com/), accessible through a [web interface](https://console.apify.com) or an [API](https://docs.apify.com/api). While we think that the Apify platform is super cool, and it's definitely worth signing up for a [free account](https://console.apify.com/sign-up), **Crawlee is and will always be open source**, runnable locally or on any cloud infrastructure. @@ -127,7 +127,7 @@ Your script will be uploaded to and built on the Apify platform so that it can b ## Usage on Apify platform -You can also develop your Actor in an online code editor directly on the platform (you'll need an Apify Account). Let's go to the [Actors](https://console.apify.com/actors) page in the app, click *Create new* and then go to the *Source* tab and start writing the code or paste one of the examples from the [Examples](../examples) section. +You can also develop your Actor in an online code editor directly on the platform (you'll need an Apify Account). Let's go to the [Actors](https://console.apify.com/actors) page in the app, click *Create new* and then go to the *Source* tab and start writing the code or paste one of the examples from the [Guides](../guides) section. ## Storages diff --git a/docs/deployment/aws_lambda.mdx b/docs/05_deployment/aws_lambda.mdx similarity index 99% rename from docs/deployment/aws_lambda.mdx rename to docs/05_deployment/aws_lambda.mdx index b07d1a2928..8d7d94123a 100644 --- a/docs/deployment/aws_lambda.mdx +++ b/docs/05_deployment/aws_lambda.mdx @@ -14,7 +14,7 @@ import PlaywrightCrawlerDockerfile from '!!raw-loader!./code_examples/aws/playwr [AWS Lambda](https://docs.aws.amazon.com/lambda/latest/dg/welcome.html) is a serverless compute service that lets you run code without provisioning or managing servers. This guide covers deploying `BeautifulSoupCrawler` and `PlaywrightCrawler`. -The code examples are based on the [BeautifulSoupCrawler example](../examples/beautifulsoup-crawler). +The code examples are based on the [BeautifulSoupCrawler example](../concepts/http-crawlers#beautifulsoupcrawler). ## BeautifulSoupCrawler on AWS Lambda diff --git a/docs/deployment/code_examples/apify/crawler_as_actor_example.py b/docs/05_deployment/code_examples/apify/crawler_as_actor_example.py similarity index 100% rename from docs/deployment/code_examples/apify/crawler_as_actor_example.py rename to docs/05_deployment/code_examples/apify/crawler_as_actor_example.py diff --git a/docs/deployment/code_examples/apify/get_public_url.py b/docs/05_deployment/code_examples/apify/get_public_url.py similarity index 100% rename from docs/deployment/code_examples/apify/get_public_url.py rename to docs/05_deployment/code_examples/apify/get_public_url.py diff --git a/docs/deployment/code_examples/apify/log_with_config_example.py b/docs/05_deployment/code_examples/apify/log_with_config_example.py similarity index 100% rename from docs/deployment/code_examples/apify/log_with_config_example.py rename to docs/05_deployment/code_examples/apify/log_with_config_example.py diff --git a/docs/deployment/code_examples/apify/proxy_advanced_example.py b/docs/05_deployment/code_examples/apify/proxy_advanced_example.py similarity index 100% rename from docs/deployment/code_examples/apify/proxy_advanced_example.py rename to docs/05_deployment/code_examples/apify/proxy_advanced_example.py diff --git a/docs/deployment/code_examples/apify/proxy_example.py b/docs/05_deployment/code_examples/apify/proxy_example.py similarity index 100% rename from docs/deployment/code_examples/apify/proxy_example.py rename to docs/05_deployment/code_examples/apify/proxy_example.py diff --git a/docs/deployment/code_examples/aws/beautifulsoup_crawler_lambda.py b/docs/05_deployment/code_examples/aws/beautifulsoup_crawler_lambda.py similarity index 100% rename from docs/deployment/code_examples/aws/beautifulsoup_crawler_lambda.py rename to docs/05_deployment/code_examples/aws/beautifulsoup_crawler_lambda.py diff --git a/docs/deployment/code_examples/aws/playwright_crawler_lambda.py b/docs/05_deployment/code_examples/aws/playwright_crawler_lambda.py similarity index 100% rename from docs/deployment/code_examples/aws/playwright_crawler_lambda.py rename to docs/05_deployment/code_examples/aws/playwright_crawler_lambda.py diff --git a/docs/deployment/code_examples/aws/playwright_dockerfile b/docs/05_deployment/code_examples/aws/playwright_dockerfile similarity index 100% rename from docs/deployment/code_examples/aws/playwright_dockerfile rename to docs/05_deployment/code_examples/aws/playwright_dockerfile diff --git a/docs/deployment/code_examples/google/cloud_run_example.py b/docs/05_deployment/code_examples/google/cloud_run_example.py similarity index 100% rename from docs/deployment/code_examples/google/cloud_run_example.py rename to docs/05_deployment/code_examples/google/cloud_run_example.py diff --git a/docs/deployment/code_examples/google/google_example.py b/docs/05_deployment/code_examples/google/google_example.py similarity index 100% rename from docs/deployment/code_examples/google/google_example.py rename to docs/05_deployment/code_examples/google/google_example.py diff --git a/docs/deployment/google_cloud.mdx b/docs/05_deployment/google_cloud.mdx similarity index 97% rename from docs/deployment/google_cloud.mdx rename to docs/05_deployment/google_cloud.mdx index e4f1fbe480..5c4a3535cf 100644 --- a/docs/deployment/google_cloud.mdx +++ b/docs/05_deployment/google_cloud.mdx @@ -14,7 +14,7 @@ import GoogleFunctions from '!!raw-loader!./code_examples/google/google_example. ## Updating the project -For the project foundation, use BeautifulSoupCrawler as described in this [example](../examples/beautifulsoup-crawler). +For the project foundation, use BeautifulSoupCrawler as described in this [example](../concepts/http-crawlers#beautifulsoupcrawler). Add [`functions-framework`](https://pypi.org/project/functions-framework/) to your dependencies file `requirements.txt`. If you're using a project manager like `poetry` or `uv`, export your dependencies to `requirements.txt`. diff --git a/docs/deployment/google_cloud_run.mdx b/docs/05_deployment/google_cloud_run.mdx similarity index 100% rename from docs/deployment/google_cloud_run.mdx rename to docs/05_deployment/google_cloud_run.mdx diff --git a/docs/upgrading/upgrading_to_v0x.md b/docs/06_upgrading/upgrading_to_v0x.md similarity index 98% rename from docs/upgrading/upgrading_to_v0x.md rename to docs/06_upgrading/upgrading_to_v0x.md index a1d861db1c..bf5558f49b 100644 --- a/docs/upgrading/upgrading_to_v0x.md +++ b/docs/06_upgrading/upgrading_to_v0x.md @@ -1,6 +1,7 @@ --- id: upgrading-to-v0x title: Upgrading to v0.x +description: Breaking changes and migration steps for upgrading between Crawlee for Python v0.x versions. --- This page summarizes the breaking changes between Crawlee for Python zero-based versions. diff --git a/docs/upgrading/upgrading_to_v1.md b/docs/06_upgrading/upgrading_to_v1.md similarity index 97% rename from docs/upgrading/upgrading_to_v1.md rename to docs/06_upgrading/upgrading_to_v1.md index 412a8919f2..ba2f37e58b 100644 --- a/docs/upgrading/upgrading_to_v1.md +++ b/docs/06_upgrading/upgrading_to_v1.md @@ -1,6 +1,7 @@ --- id: upgrading-to-v1 title: Upgrading to v1 +description: Breaking changes and migration guide for upgrading from Crawlee for Python v0.x to v1.0. --- This page summarizes the breaking changes between Crawlee for Python v0.6 and v1.0. @@ -52,13 +53,13 @@ client = HttpxHttpClient() crawler = HttpCrawler(http_client=client) ``` -See the [HTTP clients guide](https://crawlee.dev/python/docs/guides/http-clients) for all options. +See the [HTTP clients](../concepts/http-clients) page for all options. ## Changes in storages In Crawlee v1.0, the `Dataset`, `KeyValueStore`, and `RequestQueue` storage APIs have been updated for consistency and simplicity. Below is a detailed overview of what's new, what's changed, and what's been removed. -See the [Storages guide](https://crawlee.dev/python/docs/guides/storages) for more details. +See the [Storages](../concepts/storages) page for more details. ### Dataset @@ -115,7 +116,7 @@ Some changes in the related model classes: In v1.0, the storage client system has been completely reworked to simplify implementation and make custom storage clients easier to write. -See the [Storage clients guide](https://crawlee.dev/python/docs/guides/storage-clients) for more details. +See the [Storage clients](../concepts/storage-clients) page for more details. ### New dedicated storage clients diff --git a/docs/examples/add_data_to_dataset.mdx b/docs/examples/add_data_to_dataset.mdx deleted file mode 100644 index 697b157a1e..0000000000 --- a/docs/examples/add_data_to_dataset.mdx +++ /dev/null @@ -1,40 +0,0 @@ ---- -id: add-data-to-dataset -title: Add data to dataset ---- - -import ApiLink from '@site/src/components/ApiLink'; -import Tabs from '@theme/Tabs'; -import TabItem from '@theme/TabItem'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import BeautifulSoupExample from '!!raw-loader!roa-loader!./code_examples/add_data_to_dataset_bs.py'; -import PlaywrightExample from '!!raw-loader!roa-loader!./code_examples/add_data_to_dataset_pw.py'; -import DatasetExample from '!!raw-loader!roa-loader!./code_examples/add_data_to_dataset_dataset.py'; - -This example demonstrates how to store extracted data into datasets using the `context.push_data` helper function. If the specified dataset does not already exist, it will be created automatically. Additionally, you can save data to custom datasets by providing `dataset_id` or `dataset_name` parameters to the `push_data` function. - - - - - {BeautifulSoupExample} - - - - - {PlaywrightExample} - - - - -Each item in the dataset will be stored in its own file within the following directory: - -```text -{PROJECT_FOLDER}/storage/datasets/default/ -``` - -For more control, you can also open a dataset manually using the asynchronous constructor `Dataset.open` - - - {DatasetExample} - diff --git a/docs/examples/beautifulsoup_crawler.mdx b/docs/examples/beautifulsoup_crawler.mdx deleted file mode 100644 index 160e4c4d65..0000000000 --- a/docs/examples/beautifulsoup_crawler.mdx +++ /dev/null @@ -1,15 +0,0 @@ ---- -id: beautifulsoup-crawler -title: BeautifulSoup crawler ---- - -import ApiLink from '@site/src/components/ApiLink'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import BeautifulSoupExample from '!!raw-loader!roa-loader!./code_examples/beautifulsoup_crawler.py'; - -This example demonstrates how to use `BeautifulSoupCrawler` to crawl a list of URLs, load each URL using a plain HTTP request, parse the HTML using the [BeautifulSoup](https://pypi.org/project/beautifulsoup4/) library and extract some data from it - the page title and all `

`, `

` and `

` tags. This setup is perfect for scraping specific elements from web pages. Thanks to the well-known BeautifulSoup, you can easily navigate the HTML structure and retrieve the data you need with minimal code. It also shows how you can add optional pre-navigation hook to the crawler. Pre-navigation hooks are user defined functions that execute before sending the request. - - - {BeautifulSoupExample} - diff --git a/docs/examples/capture_screenshot_using_playwright.mdx b/docs/examples/capture_screenshot_using_playwright.mdx deleted file mode 100644 index 614693b1e8..0000000000 --- a/docs/examples/capture_screenshot_using_playwright.mdx +++ /dev/null @@ -1,19 +0,0 @@ ---- -id: capture-screenshots-using-playwright -title: Capture screenshots using Playwright ---- - -import ApiLink from '@site/src/components/ApiLink'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import CaptureScreenshotExample from '!!raw-loader!roa-loader!./code_examples/capture_screenshot_using_playwright.py'; - -This example demonstrates how to capture screenshots of web pages using `PlaywrightCrawler` and store them in the key-value store. - -The `PlaywrightCrawler` is configured to automate the browsing and interaction with web pages. It uses headless Chromium as the browser type to perform these tasks. Each web page specified in the initial list of URLs is visited sequentially, and a screenshot of the page is captured using Playwright's `page.screenshot()` method. - -The captured screenshots are stored in the key-value store, which is suitable for managing and storing files in various formats. In this case, screenshots are stored as PNG images with a unique key generated from the URL of the page. - - - {CaptureScreenshotExample} - diff --git a/docs/examples/capturing_page_snapshots_with_error_snapshotter.mdx b/docs/examples/capturing_page_snapshots_with_error_snapshotter.mdx deleted file mode 100644 index 87ff540298..0000000000 --- a/docs/examples/capturing_page_snapshots_with_error_snapshotter.mdx +++ /dev/null @@ -1,27 +0,0 @@ ---- -id: capturing-page-snapshots-with-error-snapshotter -title: Capturing page snapshots with ErrorSnapshotter -description: How to capture page snapshots on errors. ---- -import Tabs from '@theme/Tabs'; -import TabItem from '@theme/TabItem'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; -import ApiLink from '@site/src/components/ApiLink'; -import ParselCrawlerWithErrorSnapshotter from '!!raw-loader!roa-loader!./code_examples/parsel_crawler_with_error_snapshotter.py'; -import PlaywrightCrawlerWithErrorSnapshotter from '!!raw-loader!roa-loader!./code_examples/playwright_crawler_with_error_snapshotter.py'; - - -This example demonstrates how to capture page snapshots on first occurrence of each unique error. The capturing happens automatically if you set `save_error_snapshots=True` in the crawler's `Statistics`. The error snapshot can contain `html` file and `jpeg` file that are created from the page where the unhandled exception was raised. Captured error snapshot files are saved to the default key-value store. Both `PlaywrightCrawler` and [HTTP crawlers](../guides/http-crawlers) are capable of capturing the html file, but only `PlaywrightCrawler` is able to capture page screenshot as well. - - - - - { ParselCrawlerWithErrorSnapshotter } - - - - - { PlaywrightCrawlerWithErrorSnapshotter } - - - diff --git a/docs/examples/code_examples/add_data_to_dataset_bs.py b/docs/examples/code_examples/add_data_to_dataset_bs.py deleted file mode 100644 index 4318cbe0d4..0000000000 --- a/docs/examples/code_examples/add_data_to_dataset_bs.py +++ /dev/null @@ -1,35 +0,0 @@ -import asyncio - -from crawlee.crawlers import BeautifulSoupCrawler, BeautifulSoupCrawlingContext - - -async def main() -> None: - crawler = BeautifulSoupCrawler() - - # Define the default request handler, which will be called for every request. - @crawler.router.default_handler - async def request_handler(context: BeautifulSoupCrawlingContext) -> None: - context.log.info(f'Processing {context.request.url} ...') - - # Extract data from the page. - data = { - 'url': context.request.url, - 'title': context.soup.title.string if context.soup.title else None, - 'html': str(context.soup)[:1000], - } - - # Push the extracted data to the default dataset. - await context.push_data(data) - - # Run the crawler with the initial list of requests. - await crawler.run( - [ - 'https://crawlee.dev', - 'https://apify.com', - 'https://example.com', - ] - ) - - -if __name__ == '__main__': - asyncio.run(main()) diff --git a/docs/examples/code_examples/add_data_to_dataset_dataset.py b/docs/examples/code_examples/add_data_to_dataset_dataset.py deleted file mode 100644 index b1d9aba923..0000000000 --- a/docs/examples/code_examples/add_data_to_dataset_dataset.py +++ /dev/null @@ -1,15 +0,0 @@ -import asyncio - -from crawlee.storages import Dataset - - -async def main() -> None: - # Open dataset manually using asynchronous constructor open(). - dataset = await Dataset.open() - - # Interact with dataset directly. - await dataset.push_data({'key': 'value'}) - - -if __name__ == '__main__': - asyncio.run(main()) diff --git a/docs/examples/code_examples/add_data_to_dataset_pw.py b/docs/examples/code_examples/add_data_to_dataset_pw.py deleted file mode 100644 index 8eb714aef3..0000000000 --- a/docs/examples/code_examples/add_data_to_dataset_pw.py +++ /dev/null @@ -1,35 +0,0 @@ -import asyncio - -from crawlee.crawlers import PlaywrightCrawler, PlaywrightCrawlingContext - - -async def main() -> None: - crawler = PlaywrightCrawler() - - # Define the default request handler, which will be called for every request. - @crawler.router.default_handler - async def request_handler(context: PlaywrightCrawlingContext) -> None: - context.log.info(f'Processing {context.request.url} ...') - - # Extract data from the page. - data = { - 'url': context.request.url, - 'title': await context.page.title(), - 'html': str(await context.page.content())[:1000], - } - - # Push the extracted data to the default dataset. - await context.push_data(data) - - # Run the crawler with the initial list of requests. - await crawler.run( - [ - 'https://crawlee.dev', - 'https://apify.com', - 'https://example.com', - ] - ) - - -if __name__ == '__main__': - asyncio.run(main()) diff --git a/docs/examples/code_examples/beautifulsoup_crawler.py b/docs/examples/code_examples/beautifulsoup_crawler.py deleted file mode 100644 index 5e9701d7cb..0000000000 --- a/docs/examples/code_examples/beautifulsoup_crawler.py +++ /dev/null @@ -1,57 +0,0 @@ -import asyncio -from datetime import timedelta - -from crawlee.crawlers import ( - BasicCrawlingContext, - BeautifulSoupCrawler, - BeautifulSoupCrawlingContext, -) - - -async def main() -> None: - # Create an instance of the BeautifulSoupCrawler class, a crawler that automatically - # loads the URLs and parses their HTML using the BeautifulSoup library. - crawler = BeautifulSoupCrawler( - # On error, retry each page at most once. - max_request_retries=1, - # Increase the timeout for processing each page to 30 seconds. - request_handler_timeout=timedelta(seconds=30), - # Limit the crawl to max requests. Remove or increase it for crawling all links. - max_requests_per_crawl=10, - ) - - # Define the default request handler, which will be called for every request. - # The handler receives a context parameter, providing various properties and - # helper methods. Here are a few key ones we use for demonstration: - # - request: an instance of the Request class containing details such as the URL - # being crawled and the HTTP method used. - # - soup: the BeautifulSoup object containing the parsed HTML of the response. - @crawler.router.default_handler - async def request_handler(context: BeautifulSoupCrawlingContext) -> None: - context.log.info(f'Processing {context.request.url} ...') - - # Extract data from the page. - data = { - 'url': context.request.url, - 'title': context.soup.title.string if context.soup.title else None, - 'h1s': [h1.text for h1 in context.soup.find_all('h1')], - 'h2s': [h2.text for h2 in context.soup.find_all('h2')], - 'h3s': [h3.text for h3 in context.soup.find_all('h3')], - } - - # Push the extracted data to the default dataset. In local configuration, - # the data will be stored as JSON files in ./storage/datasets/default. - await context.push_data(data) - - # Register pre navigation hook which will be called before each request. - # This hook is optional and does not need to be defined at all. - @crawler.pre_navigation_hook - async def some_hook(context: BasicCrawlingContext) -> None: - pass - - # Run the crawler with the initial list of URLs. - await crawler.run(['https://crawlee.dev']) - - -if __name__ == '__main__': - asyncio.run(main()) diff --git a/docs/examples/code_examples/parsel_crawler.py b/docs/examples/code_examples/parsel_crawler.py deleted file mode 100644 index 9807d7ca3b..0000000000 --- a/docs/examples/code_examples/parsel_crawler.py +++ /dev/null @@ -1,47 +0,0 @@ -import asyncio - -from crawlee.crawlers import BasicCrawlingContext, ParselCrawler, ParselCrawlingContext - -# Regex for identifying email addresses on a webpage. -EMAIL_REGEX = r'[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}' - - -async def main() -> None: - crawler = ParselCrawler( - # Limit the crawl to max requests. Remove or increase it for crawling all links. - max_requests_per_crawl=10, - ) - - # Define the default request handler, which will be called for every request. - @crawler.router.default_handler - async def request_handler(context: ParselCrawlingContext) -> None: - context.log.info(f'Processing {context.request.url} ...') - - # Extract data from the page. - data = { - 'url': context.request.url, - 'title': context.selector.xpath('//title/text()').get(), - 'email_address_list': context.selector.re(EMAIL_REGEX), - } - - # Push the extracted data to the default dataset. - await context.push_data(data) - - # Enqueue all links found on the page. - await context.enqueue_links() - - # Register pre navigation hook which will be called before each request. - # This hook is optional and does not need to be defined at all. - @crawler.pre_navigation_hook - async def some_hook(context: BasicCrawlingContext) -> None: - pass - - # Run the crawler with the initial list of URLs. - await crawler.run(['https://github.com']) - - # Export the entire dataset to a JSON file. - await crawler.export_data(path='results.json') - - -if __name__ == '__main__': - asyncio.run(main()) diff --git a/docs/examples/code_examples/playwright_crawler_with_fingerprint_generator.py b/docs/examples/code_examples/playwright_crawler_with_fingerprint_generator.py deleted file mode 100644 index 24cb5bb907..0000000000 --- a/docs/examples/code_examples/playwright_crawler_with_fingerprint_generator.py +++ /dev/null @@ -1,44 +0,0 @@ -import asyncio - -from crawlee.crawlers import PlaywrightCrawler, PlaywrightCrawlingContext -from crawlee.fingerprint_suite import ( - DefaultFingerprintGenerator, - HeaderGeneratorOptions, - ScreenOptions, -) - - -async def main() -> None: - # Use default fingerprint generator with desired fingerprint options. - # Generator will generate real looking browser fingerprint based on the options. - # Unspecified fingerprint options will be automatically selected by the generator. - fingerprint_generator = DefaultFingerprintGenerator( - header_options=HeaderGeneratorOptions(browsers=['chrome']), - screen_options=ScreenOptions(min_width=400), - ) - - crawler = PlaywrightCrawler( - # Limit the crawl to max requests. Remove or increase it for crawling all links. - max_requests_per_crawl=10, - # Headless mode, set to False to see the browser in action. - headless=False, - # Browser types supported by Playwright. - browser_type='chromium', - # Fingerprint generator to be used. By default no fingerprint generation is done. - fingerprint_generator=fingerprint_generator, - ) - - # Define the default request handler, which will be called for every request. - @crawler.router.default_handler - async def request_handler(context: PlaywrightCrawlingContext) -> None: - context.log.info(f'Processing {context.request.url} ...') - - # Find a link to the next page and enqueue it if it exists. - await context.enqueue_links(selector='.morelink') - - # Run the crawler with the initial list of URLs. - await crawler.run(['https://news.ycombinator.com/']) - - -if __name__ == '__main__': - asyncio.run(main()) diff --git a/docs/examples/crawl_all_links_on_website.mdx b/docs/examples/crawl_all_links_on_website.mdx deleted file mode 100644 index f17c63920f..0000000000 --- a/docs/examples/crawl_all_links_on_website.mdx +++ /dev/null @@ -1,33 +0,0 @@ ---- -id: crawl-all-links-on-website -title: Crawl all links on website ---- - -import ApiLink from '@site/src/components/ApiLink'; -import Tabs from '@theme/Tabs'; -import TabItem from '@theme/TabItem'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import BeautifulSoupExample from '!!raw-loader!roa-loader!./code_examples/crawl_all_links_on_website_bs.py'; -import PlaywrightExample from '!!raw-loader!roa-loader!./code_examples/crawl_all_links_on_website_pw.py'; - -This example uses the `enqueue_links` helper to add new links to the `RequestQueue` as the crawler navigates from page to page. By automatically discovering and enqueuing all links on a given page, the crawler can systematically scrape an entire website. This approach is ideal for web scraping tasks where you need to collect data from multiple interconnected pages. - -:::tip - -If no options are given, by default the method will only add links that are under the same subdomain. This behavior can be controlled with the `strategy` option, which is an instance of the `EnqueueStrategy` type alias. You can find more info about this option in the [Crawl website with relative links](./crawl-website-with-relative-links) example. - -::: - - - - - {BeautifulSoupExample} - - - - - {PlaywrightExample} - - - diff --git a/docs/examples/crawl_multiple_urls.mdx b/docs/examples/crawl_multiple_urls.mdx deleted file mode 100644 index 2d3d370283..0000000000 --- a/docs/examples/crawl_multiple_urls.mdx +++ /dev/null @@ -1,27 +0,0 @@ ---- -id: crawl-multiple-urls -title: Crawl multiple URLs ---- - -import ApiLink from '@site/src/components/ApiLink'; -import Tabs from '@theme/Tabs'; -import TabItem from '@theme/TabItem'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import BeautifulSoupExample from '!!raw-loader!roa-loader!./code_examples/crawl_multiple_urls_bs.py'; -import PlaywrightExample from '!!raw-loader!roa-loader!./code_examples/crawl_multiple_urls_pw.py'; - -This example demonstrates how to crawl a specified list of URLs using different crawlers. You'll learn how to set up the crawler, define a request handler, and run the crawler with multiple URLs. This setup is useful for scraping data from multiple pages or websites concurrently. - - - - - {BeautifulSoupExample} - - - - - {PlaywrightExample} - - - diff --git a/docs/examples/crawl_specific_links_on_website.mdx b/docs/examples/crawl_specific_links_on_website.mdx deleted file mode 100644 index b350568421..0000000000 --- a/docs/examples/crawl_specific_links_on_website.mdx +++ /dev/null @@ -1,47 +0,0 @@ ---- -id: crawl-specific-links-on-website -title: Crawl specific links on website ---- - -import ApiLink from '@site/src/components/ApiLink'; -import Tabs from '@theme/Tabs'; -import TabItem from '@theme/TabItem'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import BeautifulSoupExample from '!!raw-loader!roa-loader!./code_examples/crawl_specific_links_on_website_bs.py'; -import PlaywrightExample from '!!raw-loader!roa-loader!./code_examples/crawl_specific_links_on_website_pw.py'; - -import BeautifulSoupExampleExtractAndAdd from '!!raw-loader!roa-loader!./code_examples/extract_and_add_specific_links_on_website_bs.py'; -import PlaywrightExampleExtractAndAdd from '!!raw-loader!roa-loader!./code_examples/extract_and_add_specific_links_on_website_pw.py'; - -This example demonstrates how to crawl a website while targeting specific patterns of links. By utilizing the `enqueue_links` helper, you can pass `include` or `exclude` parameters to improve your crawling strategy. This approach ensures that only the links matching the specified patterns are added to the `RequestQueue`. Both `include` and `exclude` support lists of globs or regular expressions. This functionality is great for focusing on relevant sections of a website and avoiding scraping unnecessary or irrelevant content. - - - - - {BeautifulSoupExample} - - - - - {PlaywrightExample} - - - - -## Even more control over the enqueued links - -`enqueue_links` is a convenience helper and internally it calls `extract_links` to find the links and `add_requests` to add them to the queue. If you need some additional custom filtering of the extracted links before enqueuing them, then consider using `extract_links` and `add_requests` instead of the `enqueue_links` - - - - - {BeautifulSoupExampleExtractAndAdd} - - - - - {PlaywrightExampleExtractAndAdd} - - - diff --git a/docs/examples/crawl_website_with_relative_links.mdx b/docs/examples/crawl_website_with_relative_links.mdx deleted file mode 100644 index 4cf7bee845..0000000000 --- a/docs/examples/crawl_website_with_relative_links.mdx +++ /dev/null @@ -1,52 +0,0 @@ ---- -id: crawl-website-with-relative-links -title: Crawl website with relative links ---- - -import ApiLink from '@site/src/components/ApiLink'; -import Tabs from '@theme/Tabs'; -import TabItem from '@theme/TabItem'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import AllLinksExample from '!!raw-loader!roa-loader!./code_examples/crawl_website_with_relative_links_all_links.py'; -import SameDomainExample from '!!raw-loader!roa-loader!./code_examples/crawl_website_with_relative_links_same_domain.py'; -import SameHostnameExample from '!!raw-loader!roa-loader!./code_examples/crawl_website_with_relative_links_same_hostname.py'; -import SameOriginExample from '!!raw-loader!roa-loader!./code_examples/crawl_website_with_relative_links_same_origin.py'; - -When crawling a website, you may encounter various types of links that you wish to include in your crawl. To facilitate this, we provide the `enqueue_links` method on the crawler context, which will automatically find and add these links to the crawler's `RequestQueue`. This method simplifies the process of handling different types of links, including relative links, by automatically resolving them based on the page's context. - -:::note - -For these examples, we are using the `BeautifulSoupCrawler`. However, the same method is available for other crawlers as well. You can use it in exactly the same way. - -::: - -`EnqueueStrategy` type alias provides four distinct strategies for crawling relative links: - -- `all` - Enqueues all links found, regardless of the domain they point to. This strategy is useful when you want to follow every link, including those that navigate to external websites. -- `same-domain` - Enqueues all links found that share the same domain name, including any possible subdomains. This strategy ensures that all links within the same top-level and base domain are included. -- `same-hostname` - Enqueues all links found for the exact same hostname. This is the **default** strategy, and it restricts the crawl to links that have the same hostname as the current page, excluding subdomains. -- `same-origin` - Enqueues all links found that share the same origin. The same origin refers to URLs that share the same protocol, domain, and port, ensuring a strict scope for the crawl. - - - - - {AllLinksExample} - - - - - {SameDomainExample} - - - - - {SameHostnameExample} - - - - - {SameOriginExample} - - - diff --git a/docs/examples/crawler_keep_alive.mdx b/docs/examples/crawler_keep_alive.mdx deleted file mode 100644 index 2e6c6640c7..0000000000 --- a/docs/examples/crawler_keep_alive.mdx +++ /dev/null @@ -1,15 +0,0 @@ ---- -id: crawler-keep-alive -title: Keep a Crawler alive waiting for more requests ---- - -import ApiLink from '@site/src/components/ApiLink'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import BeautifulSoupExample from '!!raw-loader!roa-loader!./code_examples/beautifulsoup_crawler_keep_alive.py'; - -This example demonstrates how to keep crawler alive even when there are no requests at the moment by using `keep_alive=True` argument of `BasicCrawler.__init__`. This is available to all crawlers that inherit from `BasicCrawler` and in the example below it is shown on `BeautifulSoupCrawler`. To stop the crawler that was started with `keep_alive=True` you can call `crawler.stop()`. - - - {BeautifulSoupExample} - diff --git a/docs/examples/crawler_stop.mdx b/docs/examples/crawler_stop.mdx deleted file mode 100644 index 4ea7f28565..0000000000 --- a/docs/examples/crawler_stop.mdx +++ /dev/null @@ -1,15 +0,0 @@ ---- -id: crawler-stop -title: Stopping a Crawler with stop method ---- - -import ApiLink from '@site/src/components/ApiLink'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import BeautifulSoupExample from '!!raw-loader!roa-loader!./code_examples/beautifulsoup_crawler_stop.py'; - -This example demonstrates how to use `stop` method of `BasicCrawler` to stop crawler once the crawler finds what it is looking for. This method is available to all crawlers that inherit from `BasicCrawler` and in the example below it is shown on `BeautifulSoupCrawler`. Simply call `crawler.stop()` to stop the crawler. It will not continue to crawl through new requests. Requests that are already being concurrently processed are going to get finished. It is possible to call `stop` method with optional argument `reason` that is a string that will be used in logs and it can improve logs readability especially if you have multiple different conditions for triggering `stop`. - - - {BeautifulSoupExample} - diff --git a/docs/examples/export_entire_dataset_to_file.mdx b/docs/examples/export_entire_dataset_to_file.mdx deleted file mode 100644 index 17c2a10a68..0000000000 --- a/docs/examples/export_entire_dataset_to_file.mdx +++ /dev/null @@ -1,47 +0,0 @@ ---- -id: export-entire-dataset-to-file -title: Export entire dataset to file ---- - -import ApiLink from '@site/src/components/ApiLink'; -import Tabs from '@theme/Tabs'; -import TabItem from '@theme/TabItem'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import JsonExample from '!!raw-loader!roa-loader!./code_examples/export_entire_dataset_to_file_json.py'; -import CsvExample from '!!raw-loader!roa-loader!./code_examples/export_entire_dataset_to_file_csv.py'; - -This example demonstrates how to use the `BasicCrawler.export_data` method of the crawler to export the entire default dataset to a single file. This method supports exporting data in either CSV or JSON format and also accepts additional keyword arguments so you can fine-tune the underlying `json.dump` or `csv.DictWriter` behavior. - -:::note - -For these examples, we are using the `BeautifulSoupCrawler`. However, the same method is available for other crawlers as well. You can use it in exactly the same way. - -::: - - - - - {JsonExample} - - - - - {CsvExample} - - - - -## CSV columns - -Dataset items don't have to share a schema. Different handlers can push different fields, so the items of one dataset often have different keys. - -By default, the CSV columns are the keys of the first non-empty item. When a later item has a key the first one doesn't, its value isn't written, and Crawlee logs a warning naming the dropped keys. To use the keys of all items as columns instead, pass `collect_all_keys=True`: - -```python -await crawler.export_data(path='results.csv', collect_all_keys=True) -``` - -No value is dropped in that mode. Collecting all keys means reading the whole dataset before the first row can be written, so only the default mode writes rows as it goes. - -In both modes, cells for columns an item doesn't have stay empty, or hold the `restval` value if you pass one. diff --git a/docs/examples/file_download.mdx b/docs/examples/file_download.mdx deleted file mode 100644 index 40c859cbe2..0000000000 --- a/docs/examples/file_download.mdx +++ /dev/null @@ -1,25 +0,0 @@ ---- -id: file-download -title: Download files ---- - -import ApiLink from '@site/src/components/ApiLink'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; -import CodeBlock from '@theme/CodeBlock'; - -import FileDownloadExample from '!!raw-loader!roa-loader!./code_examples/file_download.py'; -import FileDownloadStreamExample from '!!raw-loader!./code_examples/file_download_stream.py'; - -This example demonstrates how to use `FileDownloadCrawler` to download files such as PDFs, images or videos with plain HTTP requests. The crawler doesn't parse the response body, so any content type is accepted. Each downloaded file is saved to the default `KeyValueStore` together with the content type reported by the server. - - - {FileDownloadExample} - - -## Streaming large files - -Buffering a whole file in memory doesn't scale to large downloads. Construct the crawler with `stream=True` and the request handler receives a response whose body hasn't been read yet. Consume it in chunks with `read_stream()` and write each chunk to disk as it arrives. - - - {FileDownloadStreamExample} - diff --git a/docs/examples/json_logging.mdx b/docs/examples/json_logging.mdx deleted file mode 100644 index 06dd2ac492..0000000000 --- a/docs/examples/json_logging.mdx +++ /dev/null @@ -1,57 +0,0 @@ ---- -id: configure-json-logging -title: Сonfigure JSON logging ---- - -import ApiLink from '@site/src/components/ApiLink'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import JsonLoggingExample from '!!raw-loader!roa-loader!./code_examples/configure_json_logging.py'; - -This example demonstrates how to configure JSON line (JSONL) logging with Crawlee. By using the `use_table_logs=False` parameter, you can disable table-formatted statistics logs, which makes it easier to parse logs with external tools or to serialize them as JSON. - -The example shows how to integrate with the popular [`loguru`](https://github.com/delgan/loguru) library to capture Crawlee logs and format them as JSONL (one JSON object per line). This approach works well when you need to collect logs for analysis, monitoring, or when integrating with logging platforms like ELK Stack, Grafana Loki, or similar systems. - - - {JsonLoggingExample} - - -Here's an example of what a crawler statistics log entry in JSONL format. - -```json -{ - "text": "[HttpCrawler] | INFO | - Final request statistics: {'requests_finished': 1, 'requests_failed': 0, 'retry_histogram': [1], 'request_avg_failed_duration': None, 'request_avg_finished_duration': 3.57098, 'requests_finished_per_minute': 17, 'requests_failed_per_minute': 0, 'request_total_duration': 3.57098, 'requests_total': 1, 'crawler_runtime': 3.59165}\n", - "record": { - "elapsed": { "repr": "0:00:05.604568", "seconds": 5.604568 }, - "exception": null, - "extra": { - "requests_finished": 1, - "requests_failed": 0, - "retry_histogram": [1], - "request_avg_failed_duration": null, - "request_avg_finished_duration": 3.57098, - "requests_finished_per_minute": 17, - "requests_failed_per_minute": 0, - "request_total_duration": 3.57098, - "requests_total": 1, - "crawler_runtime": 3.59165 - }, - "file": { - "name": "_basic_crawler.py", - "path": "/crawlers/_basic/_basic_crawler.py" - }, - "function": "run", - "level": { "icon": "ℹ️", "name": "INFO", "no": 20 }, - "line": 583, - "message": "Final request statistics:", - "module": "_basic_crawler", - "name": "HttpCrawler", - "process": { "id": 198383, "name": "MainProcess" }, - "thread": { "id": 135312814966592, "name": "MainThread" }, - "time": { - "repr": "2025-03-17 17:14:45.339150+00:00", - "timestamp": 1742231685.33915 - } - } -} -``` diff --git a/docs/examples/parsel_crawler.mdx b/docs/examples/parsel_crawler.mdx deleted file mode 100644 index b0eca7eb28..0000000000 --- a/docs/examples/parsel_crawler.mdx +++ /dev/null @@ -1,15 +0,0 @@ ---- -id: parsel-crawler -title: Parsel crawler ---- - -import ApiLink from '@site/src/components/ApiLink'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import ParselCrawlerExample from '!!raw-loader!roa-loader!./code_examples/parsel_crawler.py'; - -This example shows how to use `ParselCrawler` to crawl a website or a list of URLs. Each URL is loaded using a plain HTTP request and the response is parsed using [Parsel](https://pypi.org/project/parsel/) library which supports CSS and XPath selectors for HTML responses and JMESPath for JSON responses. We can extract data from all kinds of complex HTML structures using XPath. In this example, we will use Parsel to crawl github.com and extract page title, URL and emails found in the webpage. The default handler will scrape data from the current webpage and enqueue all the links found in the webpage for continuous scraping. It also shows how you can add optional pre-navigation hook to the crawler. Pre-navigation hooks are user defined functions that execute before sending the request. - - - {ParselCrawlerExample} - diff --git a/docs/examples/playwright_crawler.mdx b/docs/examples/playwright_crawler.mdx deleted file mode 100644 index 70b0bc8afb..0000000000 --- a/docs/examples/playwright_crawler.mdx +++ /dev/null @@ -1,19 +0,0 @@ ---- -id: playwright-crawler -title: Playwright crawler ---- - -import ApiLink from '@site/src/components/ApiLink'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import PlaywrightCrawlerExample from '!!raw-loader!roa-loader!./code_examples/playwright_crawler.py'; - -This example demonstrates how to use `PlaywrightCrawler` to recursively scrape the Hacker news website using headless Chromium and Playwright. - -The `PlaywrightCrawler` manages the browser and page instances, simplifying the process of interacting with web pages. In the request handler, Playwright's API is used to extract data from each post on the page. Specifically, it retrieves the title, rank, and URL of each post. Additionally, the handler enqueues links to the next pages to ensure continuous scraping. This setup is ideal for scraping dynamic web pages where JavaScript execution is required to render the content. - -A **pre-navigation hook** can be used to perform actions before navigating to the URL. This hook provides further flexibility in controlling environment and preparing for navigation. - - - {PlaywrightCrawlerExample} - diff --git a/docs/examples/playwright_crawler_adaptive.mdx b/docs/examples/playwright_crawler_adaptive.mdx deleted file mode 100644 index f915f0246f..0000000000 --- a/docs/examples/playwright_crawler_adaptive.mdx +++ /dev/null @@ -1,20 +0,0 @@ ---- -id: adaptive-playwright-crawler -title: Adaptive Playwright crawler ---- - -import ApiLink from '@site/src/components/ApiLink'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import AdaptivePlaywrightCrawlerExample from '!!raw-loader!roa-loader!./code_examples/adaptive_playwright_crawler.py'; - -This example demonstrates how to use `AdaptivePlaywrightCrawler`. An `AdaptivePlaywrightCrawler` is a combination of `PlaywrightCrawler` and some implementation of HTTP-based crawler such as `ParselCrawler` or `BeautifulSoupCrawler`. -It uses a more limited crawling context interface so that it is able to switch to HTTP-only crawling when it detects that it may bring a performance benefit. - -A [pre-navigation hook](/python/docs/guides/adaptive-playwright-crawler#page-configuration-with-pre-navigation-hooks) can be used to perform actions before navigating to the URL. This hook provides further flexibility in controlling environment and preparing for navigation. Hooks will be executed both for the pages crawled by HTTP-bases sub crawler and playwright based sub crawler. Use `playwright_only=True` to mark hooks that should be executed only for playwright sub crawler. - -For more detailed description please see [Adaptive Playwright crawler guide](/python/docs/guides/adaptive-playwright-crawler) - - - {AdaptivePlaywrightCrawlerExample} - diff --git a/docs/examples/playwright_crawler_with_block_requests.mdx b/docs/examples/playwright_crawler_with_block_requests.mdx deleted file mode 100644 index d7d5e15928..0000000000 --- a/docs/examples/playwright_crawler_with_block_requests.mdx +++ /dev/null @@ -1,27 +0,0 @@ ---- -id: playwright-crawler-with-block-requests -title: Playwright crawler with block requests ---- - -import ApiLink from '@site/src/components/ApiLink'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import PlaywrightBlockRequests from '!!raw-loader!roa-loader!./code_examples/playwright_block_requests.py'; - -This example demonstrates how to optimize your `PlaywrightCrawler` performance by blocking unnecessary network requests. - -The primary use case is when you need to scrape or interact with web pages without loading non-essential resources like images, styles, or analytics scripts. This can significantly reduce bandwidth usage and improve crawling speed. - -The `block_requests` helper provides the most efficient way to block requests as it operates directly in the browser. - -By default, `block_requests` will block all URLs including the following patterns: - -```python -['.css', '.webp', '.jpg', '.jpeg', '.png', '.svg', '.gif', '.woff', '.pdf', '.zip'] -``` - -You can also replace the default patterns list with your own by providing `url_patterns`, or extend it by passing additional patterns in `extra_url_patterns`. - - - {PlaywrightBlockRequests} - diff --git a/docs/examples/playwright_crawler_with_camoufox.mdx b/docs/examples/playwright_crawler_with_camoufox.mdx deleted file mode 100644 index b627c9ba34..0000000000 --- a/docs/examples/playwright_crawler_with_camoufox.mdx +++ /dev/null @@ -1,26 +0,0 @@ ---- -id: playwright-crawler-with-camoufox -title: Playwright crawler with Camoufox ---- - -import ApiLink from '@site/src/components/ApiLink'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import PlaywrightCrawlerExampleWithCamoufox from '!!raw-loader!roa-loader!./code_examples/playwright_crawler_with_camoufox.py'; - -This example demonstrates how to integrate Camoufox into `PlaywrightCrawler` using `BrowserPool` with custom `PlaywrightBrowserPlugin`. - -Camoufox is a stealthy minimalistic build of Firefox. For details please visit its homepage https://camoufox.com/ . -To be able to run this example you will need to install camoufox, as it is external tool, and it is not part of the crawlee. For installation please see https://pypi.org/project/camoufox/. - -**Warning!** Camoufox is using custom build of firefox. This build can be hundreds of MB large. -You can either pre-download this file using following command `python3 -m camoufox fetch` or camoufox will download it automatically once you try to run it, and it does not find existing binary. -For more details please refer to: https://github.com/daijro/camoufox/tree/main/pythonlib#camoufox-python-interface - -**Project template -** It is possible to generate project with Python code which includes Camoufox integration into crawlee through crawlee cli. Call `crawlee create` and pick `Playwright-camoufox` when asked for Crawler type. - -The example code after PlayWrightCrawler instantiation is similar to example describing the use of Playwright Crawler. The main difference is that in this example Camoufox will be used as the browser through BrowserPool. - - - {PlaywrightCrawlerExampleWithCamoufox} - diff --git a/docs/examples/playwright_crawler_with_fingerprint_generator.mdx b/docs/examples/playwright_crawler_with_fingerprint_generator.mdx deleted file mode 100644 index 04727cd74c..0000000000 --- a/docs/examples/playwright_crawler_with_fingerprint_generator.mdx +++ /dev/null @@ -1,17 +0,0 @@ ---- -id: playwright-crawler-with-fingerprint-generator -title: Playwright crawler with fingerprint generator ---- - -import ApiLink from '@site/src/components/ApiLink'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import PlaywrightCrawlerExample from '!!raw-loader!roa-loader!./code_examples/playwright_crawler_with_fingerprint_generator.py'; - -This example demonstrates how to use `PlaywrightCrawler` together with `FingerprintGenerator` that will populate several browser attributes to mimic real browser fingerprint. To read more about fingerprints please see: https://docs.apify.com/academy/anti-scraping/techniques/fingerprinting. - -You can implement your own fingerprint generator or use `DefaultFingerprintGenerator`. To use the generator initialize it with the desired fingerprint options. The generator will try to create fingerprint based on those options. Unspecified options will be automatically selected by the generator from the set of reasonable values. If some option is important for you, do not rely on the default and explicitly define it. - - - {PlaywrightCrawlerExample} - diff --git a/docs/examples/resuming_paused_crawl.mdx b/docs/examples/resuming_paused_crawl.mdx deleted file mode 100644 index 8d2213d11d..0000000000 --- a/docs/examples/resuming_paused_crawl.mdx +++ /dev/null @@ -1,35 +0,0 @@ ---- -id: resuming-paused-crawl -title: Resuming a paused crawl ---- - -import ApiLink from '@site/src/components/ApiLink'; -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import ResumeCrawl from '!!raw-loader!roa-loader!./code_examples/resuming_paused_crawl.py'; - -This example demonstrates how to resume crawling from its last state when running locally, if for some reason it was unexpectedly terminated. - -If each run should continue crawling from the previous state, you can configure this using `purge_on_start` in `Configuration`. - -Use the code below and perform 2 sequential runs. During the 1st run, stop the crawler by pressing `CTRL+C`, and the 2nd run will resume crawling from where it stopped. - - - {ResumeCrawl} - - -Perform the 1st run, interrupting the crawler with `CTRL+C` after 2 links have been processed. - -![Run with interruption](/img/resuming-paused-crawl/00.webp 'Run with interruption.') - -Now resume crawling after the pause to process the remaining 3 links. - -![Resuming crawling](/img/resuming-paused-crawl/01.webp 'Resuming crawling.') - -Alternatively, use the environment variable `CRAWLEE_PURGE_ON_START=0` instead of using `configuration.purge_on_start = False`. - -For example, when running code: - -```bash -CRAWLEE_PURGE_ON_START=0 python -m best_crawler -``` diff --git a/docs/examples/using_browser_profile.mdx b/docs/examples/using_browser_profile.mdx deleted file mode 100644 index 8eda2554a4..0000000000 --- a/docs/examples/using_browser_profile.mdx +++ /dev/null @@ -1,39 +0,0 @@ ---- -id: using_browser_profile -title: Using browser profile ---- - -import ApiLink from '@site/src/components/ApiLink'; - -import CodeBlock from '@theme/CodeBlock'; - -import ChromeProfileExample from '!!raw-loader!./code_examples/using_browser_profiles_chrome.py'; -import FirefoxProfileExample from '!!raw-loader!./code_examples/using_browser_profiles_firefox.py'; - -This example demonstrates how to run `PlaywrightCrawler` using your local browser profile from [Chrome](https://www.google.com/intl/us/chrome/) or [Firefox](https://www.firefox.com/). - -Using browser profiles allows you to leverage existing login sessions, saved passwords, bookmarks, and other personalized browser data during crawling. This can be particularly useful for testing scenarios or when you need to access content that requires authentication. - -## Chrome browser - -To run `PlaywrightCrawler` with your Chrome profile, you need to know the path to your profile files. You can find this information by entering `chrome://version/` as a URL in your Chrome browser. If you have multiple profiles, pay attention to the profile name - if you only have one profile, it's always `Default`. - -:::warning Profile access limitation -Due to [Chrome's security policies](https://developer.chrome.com/blog/remote-debugging-port), automation cannot use your main browsing profile directly. The example copies your profile to a temporary location as a workaround. -::: - -Make sure you don't have any running Chrome browser processes before running this code: - - - {ChromeProfileExample} - - -## Firefox browser - -To find the path to your Firefox profile, enter `about:profiles` as a URL in your Firefox browser. Unlike Chrome, you can use your standard profile path directly without copying it first. - -Make sure you don't have any running Firefox browser processes before running this code: - - - {FirefoxProfileExample} - diff --git a/docs/examples/using_sitemap_request_loader.mdx b/docs/examples/using_sitemap_request_loader.mdx deleted file mode 100644 index 3ed528e94e..0000000000 --- a/docs/examples/using_sitemap_request_loader.mdx +++ /dev/null @@ -1,22 +0,0 @@ ---- -id: using-sitemap-request-loader -title: Using sitemap request loader ---- - -import ApiLink from '@site/src/components/ApiLink'; - -import RunnableCodeBlock from '@site/src/components/RunnableCodeBlock'; - -import SitemapRequestLoaderExample from '!!raw-loader!roa-loader!./code_examples/using_sitemap_request_loader.py'; - -This example demonstrates how to use `SitemapRequestLoader` to crawl websites that provide `sitemap.xml` files following the [Sitemaps protocol](https://www.sitemaps.org/protocol.html). The `SitemapRequestLoader` processes sitemaps in a streaming fashion without loading them entirely into memory, making it suitable for large sitemaps. - -The example shows how to use the `transform_request_function` parameter to configure request options based on URL patterns. This allows you to modify request properties such as labels and user data based on the source URL, enabling different handling logic for different websites or sections. - -The following code example implements processing of sitemaps from two different domains (Apify and Crawlee), with different labels assigned to requests based on their host. The `create_transform_request` function maps each host to the corresponding request configuration, while the crawler uses different handlers based on the assigned labels. - - - {SitemapRequestLoaderExample} - - -For more information about request loaders, see the [Request loaders guide](../guides/request-loaders). diff --git a/pyproject.toml b/pyproject.toml index 997fe988c5..68c7eeb5a2 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -232,16 +232,16 @@ indent-style = "space" "N999", # Invalid module name "T201", # `print` found ] -"**/docs/examples/code_examples/*crawler_with_error_snapshotter.py" = [ +"**/docs/03_concepts/code_examples/error_handling/*crawler_with_error_snapshotter.py" = [ "PLR2004", # Magic value used in comparison. Ignored for simplicity and readability of example code. ] -"**/docs/guides/code_examples/running_in_web_server/server.py" = [ +"**/docs/04_guides/code_examples/running_in_web_server/server.py" = [ "TC002", # ruff false positive. Import actually needed during runtime. ] -"**/docs/guides/code_examples/creating_web_archive/*.*" = [ +"**/docs/04_guides/code_examples/creating_web_archive/*.*" = [ "ASYNC230", # Ignore for simplicity of the example. ] -"**/docs/guides/code_examples/scrapy_migration/scrapy_*.py" = [ +"**/docs/04_guides/code_examples/scrapy_migration/scrapy_*.py" = [ "ANN", # Idiomatic Scrapy tutorial snippets are untyped; shown for comparison, not executed. "RUF012", # Scrapy declares `start_urls` as a plain class-level list. ] @@ -285,7 +285,7 @@ python-version = "3.10" include = ["src", "tests", "scripts", "docs", "website"] exclude = [ "src/crawlee/project_template", - "docs/guides/code_examples/storage_clients/custom_storage_client_example.py", + "docs/03_concepts/code_examples/storage_clients/custom_storage_client_example.py", "website/versioned_docs", ] diff --git a/website/docusaurus.config.js b/website/docusaurus.config.js index 34d243e8a9..9fff82d8df 100644 --- a/website/docusaurus.config.js +++ b/website/docusaurus.config.js @@ -120,40 +120,112 @@ module.exports = { }, }, ], - // [ - // '@docusaurus/plugin-client-redirects', - // { - // redirects: [ - // { - // from: '/docs', - // to: '/docs/quick-start', - // }, - // { - // from: '/docs/next', - // to: '/docs/next/quick-start', - // }, - // { - // from: '/docs/guides/environment-variables', - // to: '/docs/guides/configuration', - // }, - // { - // from: '/docs/guides/getting-started', - // to: '/docs/introduction', - // }, - // { - // from: '/docs/guides/apify-platform', - // to: '/docs/deployment/apify-platform', - // }, - // ], - // createRedirects(existingPath) { - // if (!existingPath.endsWith('/')) { - // return `${existingPath}/`; - // } - // - // return undefined; // Return a falsy value: no redirect created - // }, - // }, - // ], + [ + '@docusaurus/plugin-client-redirects', + { + createRedirects(existingPath) { + // Maps a current docs path suffix to the old suffixes that should redirect + // to it, covering the restructuring that introduced the Concepts section + // and merged the Examples section into Concepts and Guides. Redirects are + // derived from existing routes, so they apply to every docs version that + // contains the new page and never shadow versions that still have the old + // layout (see OLD_STRUCTURE_VERSIONS below). + const MOVED_DOCS = { + 'concepts/architecture-overview': ['guides/architecture-overview'], + 'concepts/http-crawlers': [ + 'guides/http-crawlers', + 'examples/beautifulsoup-crawler', + 'examples/parsel-crawler', + 'examples/file-download', + ], + 'concepts/playwright-crawler': [ + 'guides/playwright-crawler', + 'examples/playwright-crawler', + 'examples/playwright-crawler-with-block-requests', + 'examples/capture-screenshots-using-playwright', + ], + 'concepts/adaptive-playwright-crawler': [ + 'guides/adaptive-playwright-crawler', + 'examples/adaptive-playwright-crawler', + ], + 'concepts/request-router': ['guides/request-router'], + 'concepts/request-loaders': ['guides/request-loaders', 'examples/using-sitemap-request-loader'], + 'concepts/storages': [ + 'guides/storages', + 'examples/add-data-to-dataset', + 'examples/export-entire-dataset-to-file', + ], + 'concepts/storage-clients': ['guides/storage-clients'], + 'concepts/http-clients': ['guides/http-clients'], + 'concepts/session-management': ['guides/session-management'], + 'concepts/cookie-management': ['guides/cookie-management'], + 'concepts/http-headers': ['guides/http-headers'], + 'concepts/proxy-management': ['guides/proxy-management'], + 'concepts/scaling-crawlers': ['guides/scaling-crawlers'], + 'concepts/request-throttling': ['guides/request-throttling'], + 'concepts/error-handling': [ + 'guides/error-handling', + 'examples/capturing-page-snapshots-with-error-snapshotter', + ], + 'concepts/logging': ['examples/configure-json-logging'], + 'concepts/service-locator': ['guides/service-locator'], + 'guides/crawling-links': [ + 'examples/crawl-all-links-on-website', + 'examples/crawl-multiple-urls', + 'examples/crawl-specific-links-on-website', + 'examples/crawl-website-with-relative-links', + ], + 'guides/fill-and-submit-web-form': ['examples/fill-and-submit-web-form'], + 'guides/respect-robots-txt-file': ['examples/respect-robots-txt-file'], + 'guides/stopping-and-resuming-crawlers': [ + 'examples/crawler-stop', + 'examples/crawler-keep-alive', + 'examples/resuming-paused-crawl', + ], + 'guides/run-parallel-crawlers': ['examples/run-parallel-crawlers'], + 'guides/avoid-blocking': [ + 'examples/playwright-crawler-with-camoufox', + 'examples/playwright-crawler-with-fingerprint-generator', + ], + 'guides/logging-in-with-a-crawler': ['examples/using_browser_profile'], + guides: ['examples'], + }; + + // Doc versions that still have the pre-restructure layout. Their routes + // (e.g. /docs/1.9/examples/...) must not get redirects generated over them. + const OLD_STRUCTURE_VERSIONS = ['0.6', '1.9']; + const versions = require('./versions.json'); + const latestHasOldStructure = OLD_STRUCTURE_VERSIONS.includes(versions[0]); + + const marker = '/docs/'; + const markerIndex = existingPath.indexOf(marker); + + if (markerIndex === -1) { + return undefined; + } + + const redirects = []; + + for (const [currentSuffix, oldSuffixes] of Object.entries(MOVED_DOCS)) { + if (!existingPath.endsWith(`/${currentSuffix}`)) { + continue; + } + + // E.g. '/python/docs/', '/python/docs/next/' or '/python/docs/1.9/'. + const docsRoot = existingPath.slice(0, existingPath.length - currentSuffix.length); + const version = docsRoot.slice(markerIndex + marker.length).replace(/\/$/, ''); + const hasOldStructure = + version === '' ? latestHasOldStructure : OLD_STRUCTURE_VERSIONS.includes(version); + + if (!hasOldStructure) { + redirects.push(...oldSuffixes.map((oldSuffix) => `${docsRoot}${oldSuffix}`)); + } + } + + return redirects.length > 0 ? redirects : undefined; + }, + }, + ], [ 'docusaurus-gtm-plugin', { @@ -288,12 +360,6 @@ module.exports = { label: 'Docs', position: 'left', }, - { - type: 'doc', - docId: '/examples', - label: 'Examples', - position: 'left', - }, { type: 'custom-api', label: 'API', @@ -351,12 +417,12 @@ module.exports = { title: 'Docs', items: [ { - label: 'Guides', - to: 'docs/guides', + label: 'Quick start', + to: 'docs/quick-start', }, { - label: 'Examples', - to: 'docs/examples', + label: 'Guides', + to: 'docs/guides', }, { label: 'API reference', diff --git a/website/sidebars.js b/website/sidebars.js index 8379abf15f..cab87b4881 100644 --- a/website/sidebars.js +++ b/website/sidebars.js @@ -21,6 +21,25 @@ module.exports = { 'introduction/deployment', ], }, + { + type: 'category', + label: 'Concepts', + collapsed: true, + link: { + type: 'generated-index', + title: 'Concepts', + description: + 'Learn how the core components of Crawlee work - crawlers, request routing, storages, sessions, and more.', + slug: '/concepts', + keywords: ['concepts'], + }, + items: [ + { + type: 'autogenerated', + dirName: '03_concepts', + }, + ], + }, { type: 'category', label: 'Guides', @@ -28,13 +47,15 @@ module.exports = { link: { type: 'generated-index', title: 'Guides', + description: + 'Task-oriented guides for common crawling and scraping scenarios, integrations, and migrations.', slug: '/guides', keywords: ['guides'], }, items: [ { type: 'autogenerated', - dirName: 'guides', + dirName: '04_guides', }, ], }, @@ -66,39 +87,6 @@ module.exports = { }, ], }, - { - type: 'category', - label: 'Examples', - collapsed: true, - link: { - type: 'generated-index', - title: 'Examples', - slug: '/examples', - keywords: ['examples'], - }, - items: [ - { - type: 'autogenerated', - dirName: 'examples', - }, - ], - }, - // { - // type: 'category', - // label: 'Experiments', - // link: { - // type: 'generated-index', - // title: 'Experiments', - // slug: '/experiments', - // keywords: ['experiments', 'experimental-features'], - // }, - // items: [ - // { - // type: 'autogenerated', - // dirName: 'experiments', - // }, - // ], - // }, { type: 'category', label: 'Upgrading', @@ -112,7 +100,7 @@ module.exports = { items: [ { type: 'autogenerated', - dirName: 'upgrading', + dirName: '06_upgrading', }, ], }, diff --git a/website/src/pages/index.js b/website/src/pages/index.js index b4c6d58ccd..1867602759 100644 --- a/website/src/pages/index.js +++ b/website/src/pages/index.js @@ -107,13 +107,15 @@ function BenefitsSection() { ); } +// The /docs/next prefix on the concepts links is temporary: the Concepts section exists only in the +// next (unreleased) docs until the first post-restructure version snapshot. Drop the prefix then. function OtherFeaturesSection() { return (

What else is in Crawlee?

- +
- +