[{"id": "5cb78ebfcb44f21e", "url": "https://docs.crawl4ai.com", "page_title": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper", "page_type": "overview", "page_summary": "Crawl4AI is an open-source, LLM-friendly web crawler and scraper that produces clean Markdown, supports structured extraction, advanced browser control, and high-performance parallel crawling. This…", "heading": "🚀 Crawl4AI Cloud API — Closed Beta (Launching Soon)", "content": "Page: 🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper\nSection: 🚀 Crawl4AI Cloud API — Closed Beta (Launching Soon)\n\nReliable, large-scale web extraction, now built to be **drastically more cost-effective** than any of the existing solutions.\n\n👉 **Apply [here](https://forms.gle/E9MyPaNXACnAMaqG7) for early…", "code_blocks": [], "chunk_position": 0, "heading_path": "🚀 Crawl4AI Cloud API — Closed Beta (Launching Soon) > 🚀 Crawl4AI Cloud API — Closed Beta (Launching Soon)", "breadcrumbs": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper > 🚀 Crawl4AI Cloud API — Closed Beta (Launching Soon) > 🚀 Crawl4AI Cloud API — Closed Beta (Launching Soon)"}, {"id": "b72f943bb078ba7e", "url": "https://docs.crawl4ai.com", "page_title": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper", "page_type": "overview", "page_summary": "Crawl4AI is an open-source, LLM-friendly web crawler and scraper that produces clean Markdown, supports structured extraction, advanced browser control, and high-performance parallel crawling. This…", "heading": "Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper", "content": "Page: 🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper\nSection: Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper\n\nCrawl4AI is the #1 trending GitHub repository, actively maintained by a vibrant community. It delivers blazing-fast, AI-ready web crawling tailored for large language models, AI agents, and data…", "code_blocks": [], "chunk_position": 0, "heading_path": "Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper > Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper", "breadcrumbs": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper > Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper > Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper"}, {"id": "f91d22f055199b7a", "url": "https://docs.crawl4ai.com", "page_title": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper", "page_type": "overview", "page_summary": "Crawl4AI is an open-source, LLM-friendly web crawler and scraper that produces clean Markdown, supports structured extraction, advanced browser control, and high-performance parallel crawling. This…", "heading": "🤖 Crawl4AI Skill for Claude & AI Assistants", "content": "Page: 🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper\nSection: 🤖 Crawl4AI Skill for Claude & AI Assistants\n\nSupercharge your AI coding assistant with complete Crawl4AI knowledge! Download our comprehensive skill package that includes:\n\n- 📚 Complete SDK reference (23K+ words)\n- 🚀 Ready-to-use extraction…", "code_blocks": [], "chunk_position": 0, "heading_path": "🤖 Crawl4AI Skill for Claude & AI Assistants > 🤖 Crawl4AI Skill for Claude & AI Assistants", "breadcrumbs": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper > 🤖 Crawl4AI Skill for Claude & AI Assistants > 🤖 Crawl4AI Skill for Claude & AI Assistants"}, {"id": "a6bed72a9b64d2b9", "url": "https://docs.crawl4ai.com", "page_title": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper", "page_type": "overview", "page_summary": "Crawl4AI is an open-source, LLM-friendly web crawler and scraper that produces clean Markdown, supports structured extraction, advanced browser control, and high-performance parallel crawling. This…", "heading": "🎯 New: Adaptive Web Crawling", "content": "Page: 🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper\nSection: 🎯 New: Adaptive Web Crawling\n\nCrawl4AI now features intelligent adaptive crawling that knows when to stop! Using advanced information foraging algorithms, it determines when sufficient information has been gathered to answer your…", "code_blocks": [], "chunk_position": 0, "heading_path": "🎯 New: Adaptive Web Crawling > 🎯 New: Adaptive Web Crawling", "breadcrumbs": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper > 🎯 New: Adaptive Web Crawling > 🎯 New: Adaptive Web Crawling"}, {"id": "3e90ee88b9d1cf83", "url": "https://docs.crawl4ai.com", "page_title": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper", "page_type": "overview", "page_summary": "Crawl4AI is an open-source, LLM-friendly web crawler and scraper that produces clean Markdown, supports structured extraction, advanced browser control, and high-performance parallel crawling. This…", "heading": "Quick Start", "content": "Page: 🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper\nSection: Quick Start\n\nHere's a quick example to show you how easy it is to use Crawl4AI with its asynchronous capabilities:", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler\n\nasync def main():\n    # Create an instance of AsyncWebCrawler\n    async with AsyncWebCrawler() as crawler:\n        # Run the crawler on a URL…", "filename": ""}], "chunk_position": 0, "heading_path": "Quick Start > Quick Start", "breadcrumbs": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper > Quick Start > Quick Start"}, {"id": "b4c3d9b21315888a", "url": "https://docs.crawl4ai.com", "page_title": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper", "page_type": "overview", "page_summary": "Crawl4AI is an open-source, LLM-friendly web crawler and scraper that produces clean Markdown, supports structured extraction, advanced browser control, and high-performance parallel crawling. This…", "heading": "What Does Crawl4AI Do?", "content": "Page: 🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper\nSection: What Does Crawl4AI Do?\n\nCrawl4AI is a feature-rich crawler and scraper that aims to:\n\n1. **Generate Clean Markdown**: Perfect for RAG pipelines or direct ingestion into LLMs.\n2. **Structured Extraction**: Parse repeated…", "code_blocks": [], "chunk_position": 0, "heading_path": "What Does Crawl4AI Do? > What Does Crawl4AI Do?", "breadcrumbs": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper > What Does Crawl4AI Do? > What Does Crawl4AI Do?"}, {"id": "f064a630a3e8a2c6", "url": "https://docs.crawl4ai.com", "page_title": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper", "page_type": "overview", "page_summary": "Crawl4AI is an open-source, LLM-friendly web crawler and scraper that produces clean Markdown, supports structured extraction, advanced browser control, and high-performance parallel crawling. This…", "heading": "Documentation Structure", "content": "Page: 🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper\nSection: Documentation Structure\n\nTo help you get started, we’ve organized our docs into clear sections:\n\n- **Setup & Installation**: Basic instructions to install Crawl4AI via pip or Docker.\n- **Quick Start**: A hands-on…", "code_blocks": [], "chunk_position": 0, "heading_path": "Documentation Structure > Documentation Structure", "breadcrumbs": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper > Documentation Structure > Documentation Structure"}, {"id": "0409c6fc001d4e68", "url": "https://docs.crawl4ai.com", "page_title": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper", "page_type": "overview", "page_summary": "Crawl4AI is an open-source, LLM-friendly web crawler and scraper that produces clean Markdown, supports structured extraction, advanced browser control, and high-performance parallel crawling. This…", "heading": "How You Can Support", "content": "Page: 🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper\nSection: How You Can Support\n\n- **Star & Fork**: If you find Crawl4AI helpful, star the repo on GitHub or fork it to add your own features.\n- **File Issues**: Encounter a bug or missing feature? Let us know by filing an issue, so…", "code_blocks": [], "chunk_position": 0, "heading_path": "How You Can Support > How You Can Support", "breadcrumbs": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper > How You Can Support > How You Can Support"}, {"id": "53b3b356635460dc", "url": "https://docs.crawl4ai.com", "page_title": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper", "page_type": "overview", "page_summary": "Crawl4AI is an open-source, LLM-friendly web crawler and scraper that produces clean Markdown, supports structured extraction, advanced browser control, and high-performance parallel crawling. This…", "heading": "Quick Links", "content": "Page: 🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper\nSection: Quick Links\n\n- **[GitHub Repo](https://github.com/unclecode/crawl4ai)**\n- **[Installation Guide](core/installation/)**\n- **[Quick Start](core/quickstart/)**\n- **[API Reference](api/async-webcrawler/)**\n-…", "code_blocks": [], "chunk_position": 0, "heading_path": "Quick Links > Quick Links", "breadcrumbs": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper > Quick Links > Quick Links"}, {"id": "f5ff927def6b235b", "url": "https://docs.crawl4ai.com/", "page_title": "Home - Crawl4AI Documentation (v0.9.x)", "page_type": "overview", "page_summary": "Landing page and overview for Crawl4AI, an open-source, LLM-friendly web crawler and scraper. It introduces the project's mission, key capabilities, quick-start example, documentation structure, and…", "heading": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper", "content": "Page: Home - Crawl4AI Documentation (v0.9.x)\nSection: 🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper\n\n🚀 Crawl4AI Cloud API — Closed Beta (Launching Soon)\n\nReliable, large-scale web extraction, now built to be drastically more cost-effective than any of the existing solutions.\n\n👉 Apply here for early…", "code_blocks": [], "chunk_position": 1, "heading_path": "🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper > 🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper", "breadcrumbs": "Home - Crawl4AI Documentation (v0.9.x) > 🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper > 🚀🤖 Crawl4AI: Open-Source LLM-Friendly Web Crawler & Scraper"}, {"id": "0f57e70813d6d45d", "url": "https://docs.crawl4ai.com/", "page_title": "Home - Crawl4AI Documentation (v0.9.x)", "page_type": "overview", "page_summary": "Landing page and overview for Crawl4AI, an open-source, LLM-friendly web crawler and scraper. It introduces the project's mission, key capabilities, quick-start example, documentation structure, and…", "heading": "🆕 AI Assistant Skill Now Available!", "content": "Page: Home - Crawl4AI Documentation (v0.9.x)\nSection: 🆕 AI Assistant Skill Now Available!\n\n🤖 Crawl4AI Skill for Claude & AI Assistants\n\nSupercharge your AI coding assistant with complete Crawl4AI knowledge! Download our comprehensive skill package that includes:\n\n- 📚 Complete SDK reference…", "code_blocks": [], "chunk_position": 1, "heading_path": "🆕 AI Assistant Skill Now Available! > 🆕 AI Assistant Skill Now Available!", "breadcrumbs": "Home - Crawl4AI Documentation (v0.9.x) > 🆕 AI Assistant Skill Now Available! > 🆕 AI Assistant Skill Now Available!"}, {"id": "3c14ba343eb1475a", "url": "https://docs.crawl4ai.com/", "page_title": "Home - Crawl4AI Documentation (v0.9.x)", "page_type": "overview", "page_summary": "Landing page and overview for Crawl4AI, an open-source, LLM-friendly web crawler and scraper. It introduces the project's mission, key capabilities, quick-start example, documentation structure, and…", "heading": "🎯 New: Adaptive Web Crawling", "content": "Page: Home - Crawl4AI Documentation (v0.9.x)\nSection: 🎯 New: Adaptive Web Crawling\n\nCrawl4AI now features intelligent adaptive crawling that knows when to stop! Using advanced information foraging algorithms, it determines when sufficient information has been gathered to answer your…", "code_blocks": [], "chunk_position": 1, "heading_path": "🎯 New: Adaptive Web Crawling > 🎯 New: Adaptive Web Crawling", "breadcrumbs": "Home - Crawl4AI Documentation (v0.9.x) > 🎯 New: Adaptive Web Crawling > 🎯 New: Adaptive Web Crawling"}, {"id": "0ea7346502f94054", "url": "https://docs.crawl4ai.com/", "page_title": "Home - Crawl4AI Documentation (v0.9.x)", "page_type": "overview", "page_summary": "Landing page and overview for Crawl4AI, an open-source, LLM-friendly web crawler and scraper. It introduces the project's mission, key capabilities, quick-start example, documentation structure, and…", "heading": "Quick Start", "content": "Page: Home - Crawl4AI Documentation (v0.9.x)\nSection: Quick Start\n\nHere's a quick example to show you how easy it is to use Crawl4AI with its asynchronous capabilities:", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler\n\nasync def main():\n    # Create an instance of AsyncWebCrawler\n    async with AsyncWebCrawler() as crawler:\n        # Run the crawler on a URL…", "filename": ""}], "chunk_position": 1, "heading_path": "Quick Start > Quick Start", "breadcrumbs": "Home - Crawl4AI Documentation (v0.9.x) > Quick Start > Quick Start"}, {"id": "df32d0b9121918cd", "url": "https://docs.crawl4ai.com/", "page_title": "Home - Crawl4AI Documentation (v0.9.x)", "page_type": "overview", "page_summary": "Landing page and overview for Crawl4AI, an open-source, LLM-friendly web crawler and scraper. It introduces the project's mission, key capabilities, quick-start example, documentation structure, and…", "heading": "What Does Crawl4AI Do?", "content": "Page: Home - Crawl4AI Documentation (v0.9.x)\nSection: What Does Crawl4AI Do?\n\nCrawl4AI is a feature-rich crawler and scraper that aims to:\n\n1. Generate Clean Markdown: Perfect for RAG pipelines or direct ingestion into LLMs.\n2. Structured Extraction: Parse repeated patterns…", "code_blocks": [], "chunk_position": 1, "heading_path": "What Does Crawl4AI Do? > What Does Crawl4AI Do?", "breadcrumbs": "Home - Crawl4AI Documentation (v0.9.x) > What Does Crawl4AI Do? > What Does Crawl4AI Do?"}, {"id": "d2c8780548719746", "url": "https://docs.crawl4ai.com/", "page_title": "Home - Crawl4AI Documentation (v0.9.x)", "page_type": "overview", "page_summary": "Landing page and overview for Crawl4AI, an open-source, LLM-friendly web crawler and scraper. It introduces the project's mission, key capabilities, quick-start example, documentation structure, and…", "heading": "Documentation Structure", "content": "Page: Home - Crawl4AI Documentation (v0.9.x)\nSection: Documentation Structure\n\nTo help you get started, we’ve organized our docs into clear sections:\n\n- Setup & Installation — Basic instructions to install Crawl4AI via pip or Docker.\n- Quick Start — A hands-on introduction…", "code_blocks": [], "chunk_position": 1, "heading_path": "Documentation Structure > Documentation Structure", "breadcrumbs": "Home - Crawl4AI Documentation (v0.9.x) > Documentation Structure > Documentation Structure"}, {"id": "c03288f25e76f65e", "url": "https://docs.crawl4ai.com/", "page_title": "Home - Crawl4AI Documentation (v0.9.x)", "page_type": "overview", "page_summary": "Landing page and overview for Crawl4AI, an open-source, LLM-friendly web crawler and scraper. It introduces the project's mission, key capabilities, quick-start example, documentation structure, and…", "heading": "How You Can Support", "content": "Page: Home - Crawl4AI Documentation (v0.9.x)\nSection: How You Can Support\n\n- Star & Fork: If you find Crawl4AI helpful, star the repo on GitHub or fork it to add your own features.\n- File Issues: Encounter a bug or missing feature? Let us know by filing an issue, so we can…", "code_blocks": [], "chunk_position": 1, "heading_path": "How You Can Support > How You Can Support", "breadcrumbs": "Home - Crawl4AI Documentation (v0.9.x) > How You Can Support > How You Can Support"}, {"id": "b363a8ce4eb955fe", "url": "https://docs.crawl4ai.com/", "page_title": "Home - Crawl4AI Documentation (v0.9.x)", "page_type": "overview", "page_summary": "Landing page and overview for Crawl4AI, an open-source, LLM-friendly web crawler and scraper. It introduces the project's mission, key capabilities, quick-start example, documentation structure, and…", "heading": "Quick Links", "content": "Page: Home - Crawl4AI Documentation (v0.9.x)\nSection: Quick Links\n\n- GitHub Repo\n- Installation Guide\n- Quick Start\n- API Reference\n- Changelog\n\nThank you for joining me on this journey. Let’s keep building an open, democratic approach to data extraction and AI…", "code_blocks": [], "chunk_position": 1, "heading_path": "Quick Links > Quick Links", "breadcrumbs": "Home - Crawl4AI Documentation (v0.9.x) > Quick Links > Quick Links"}, {"id": "c73a859d5f638c74", "url": "https://docs.crawl4ai.com/CONTRIBUTING/", "page_title": "Contributing Guide - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "Contribution guide for Crawl4AI explaining the branching strategy, contributor workflow, release process, and best practices for submitting pull requests.", "heading": "Introduction", "content": "Page: Contributing Guide - Crawl4AI Documentation (v0.9.x)\nSection: Introduction\n\nWelcome to the Crawl4AI project! As an open-source library for web crawling and AI integration, we value contributions from the community. This guide explains our branching strategy, how to…", "code_blocks": [], "chunk_position": 2, "heading_path": "Introduction > Introduction", "breadcrumbs": "Contributing Guide - Crawl4AI Documentation (v0.9.x) > Introduction > Introduction"}, {"id": "cb3e77c3a78518ae", "url": "https://docs.crawl4ai.com/CONTRIBUTING/", "page_title": "Contributing Guide - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "Contribution guide for Crawl4AI explaining the branching strategy, contributor workflow, release process, and best practices for submitting pull requests.", "heading": "Core Branches", "content": "Page: Contributing Guide - Crawl4AI Documentation (v0.9.x)\nSection: Core Branches\n\n- **main** : The stable branch containing production-ready code. It's always identical to the latest released version and is tagged for releases. Do not submit PRs directly here.\n\n- **develop** : The…", "code_blocks": [], "chunk_position": 2, "heading_path": "Core Branches > Core Branches", "breadcrumbs": "Contributing Guide - Crawl4AI Documentation (v0.9.x) > Core Branches > Core Branches"}, {"id": "e5f88439f9871edb", "url": "https://docs.crawl4ai.com/CONTRIBUTING/", "page_title": "Contributing Guide - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "Contribution guide for Crawl4AI explaining the branching strategy, contributor workflow, release process, and best practices for submitting pull requests.", "heading": "Contributor Workflow", "content": "Page: Contributing Guide - Crawl4AI Documentation (v0.9.x)\nSection: Contributor Workflow\n\nWe encourage contributions of all kinds: bug fixes, new features, documentation improvements, tests, or even Docker enhancements. Follow these steps to contribute:\n\n- **Fork the Repository** : Create…", "code_blocks": [{"language": "shell", "code": "git checkout develop\ngit checkout -b feature/your-feature-name  # Or bugfix/your-bugfix-name", "filename": ""}], "chunk_position": 2, "heading_path": "Contributor Workflow > Contributor Workflow", "breadcrumbs": "Contributing Guide - Crawl4AI Documentation (v0.9.x) > Contributor Workflow > Contributor Workflow"}, {"id": "f7d863266859d590", "url": "https://docs.crawl4ai.com/CONTRIBUTING/", "page_title": "Contributing Guide - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "Contribution guide for Crawl4AI explaining the branching strategy, contributor workflow, release process, and best practices for submitting pull requests.", "heading": "Lead Maintainer's Workflow (For Reference)", "content": "Page: Contributing Guide - Crawl4AI Documentation (v0.9.x)\nSection: Lead Maintainer's Workflow (For Reference)\n\n- The lead maintainer (Unclecode) uses the `next` branch for isolated experimental work.\n\n- Features from `next` are periodically merged into `develop` (via rebase and merge) to keep everything in…", "code_blocks": [], "chunk_position": 2, "heading_path": "Lead Maintainer's Workflow (For Reference) > Lead Maintainer's Workflow (For Reference)", "breadcrumbs": "Contributing Guide - Crawl4AI Documentation (v0.9.x) > Lead Maintainer's Workflow (For Reference) > Lead Maintainer's Workflow (For Reference)"}, {"id": "3904b7cb0e91630c", "url": "https://docs.crawl4ai.com/CONTRIBUTING/", "page_title": "Contributing Guide - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "Contribution guide for Crawl4AI explaining the branching strategy, contributor workflow, release process, and best practices for submitting pull requests.", "heading": "Release Process (High-Level Overview)", "content": "Page: Contributing Guide - Crawl4AI Documentation (v0.9.x)\nSection: Release Process (High-Level Overview)\n\nReleases happen bi-weekly to ship improvements regularly. As a contributor, your merged changes in `develop` will be included in the next release unless specified otherwise. Here's a summary of what…", "code_blocks": [], "chunk_position": 2, "heading_path": "Release Process (High-Level Overview) > Release Process (High-Level Overview)", "breadcrumbs": "Contributing Guide - Crawl4AI Documentation (v0.9.x) > Release Process (High-Level Overview) > Release Process (High-Level Overview)"}, {"id": "fc089ee37e82234b", "url": "https://docs.crawl4ai.com/CONTRIBUTING/", "page_title": "Contributing Guide - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "Contribution guide for Crawl4AI explaining the branching strategy, contributor workflow, release process, and best practices for submitting pull requests.", "heading": "Benefits of This Approach", "content": "Page: Contributing Guide - Crawl4AI Documentation (v0.9.x)\nSection: Benefits of This Approach\n\n- **Stability** : `main` is always reliable for users.\n\n- **Collaboration** : Fixed PR target (`develop`) makes contributing straightforward.\n\n- **Isolation** : Experimental work in `next` doesn't…", "code_blocks": [], "chunk_position": 2, "heading_path": "Benefits of This Approach > Benefits of This Approach", "breadcrumbs": "Contributing Guide - Crawl4AI Documentation (v0.9.x) > Benefits of This Approach > Benefits of This Approach"}, {"id": "a3f59bf6cc747e1a", "url": "https://docs.crawl4ai.com/CONTRIBUTING/", "page_title": "Contributing Guide - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "Contribution guide for Crawl4AI explaining the branching strategy, contributor workflow, release process, and best practices for submitting pull requests.", "heading": "Checklist for Contributors", "content": "Page: Contributing Guide - Crawl4AI Documentation (v0.9.x)\nSection: Checklist for Contributors\n\nBefore submitting a PR:\n\n- [ ]  Based on and targeting `develop`.\n\n- [ ]  Tests pass (`pytest`).\n\n- [ ]  Docs updated if needed (e.g., version refs in mkdocs.yml, Docker files).\n\n- [ ]  No breaking…", "code_blocks": [], "chunk_position": 2, "heading_path": "Checklist for Contributors > Checklist for Contributors", "breadcrumbs": "Contributing Guide - Crawl4AI Documentation (v0.9.x) > Checklist for Contributors > Checklist for Contributors"}, {"id": "990b84abf24d025c", "url": "https://docs.crawl4ai.com/CONTRIBUTING/", "page_title": "Contributing Guide - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "Contribution guide for Crawl4AI explaining the branching strategy, contributor workflow, release process, and best practices for submitting pull requests.", "heading": "Common Issues", "content": "Page: Contributing Guide - Crawl4AI Documentation (v0.9.x)\nSection: Common Issues\n\n- **Merge Conflicts** : Rebase your branch on latest `develop` before PR.\n\n- **Docker Builds** : Test multi-arch (amd64/arm64) locally if changing Dockerfile.\n\n- **Version Consistency** : Ensure any…", "code_blocks": [], "chunk_position": 2, "heading_path": "Common Issues > Common Issues", "breadcrumbs": "Contributing Guide - Crawl4AI Documentation (v0.9.x) > Common Issues > Common Issues"}, {"id": "695350e241db593d", "url": "https://docs.crawl4ai.com/CONTRIBUTING/", "page_title": "Contributing Guide - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "Contribution guide for Crawl4AI explaining the branching strategy, contributor workflow, release process, and best practices for submitting pull requests.", "heading": "Communication", "content": "Page: Contributing Guide - Crawl4AI Documentation (v0.9.x)\nSection: Communication\n\n- Open issues for discussions or bugs.\n\n- Join our Discord (link in README) for real-time help.\n\n- After releases, announcements go to GitHub, Discord, and social media.\n\nThanks for contributing to…", "code_blocks": [], "chunk_position": 2, "heading_path": "Communication > Communication", "breadcrumbs": "Contributing Guide - Crawl4AI Documentation (v0.9.x) > Communication > Communication"}, {"id": "ebcd4d1c35793cc1", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "Overview", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Overview\n\nWhile the default adaptive crawling configuration works well for most use cases, understanding the underlying strategies and scoring mechanisms allows you to fine-tune the crawler for specific…", "code_blocks": [], "chunk_position": 3, "heading_path": "Overview > Overview", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > Overview > Overview"}, {"id": "46652ed374d17910", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "1. Coverage Score", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: 1. Coverage Score\n\nCoverage measures how comprehensively your knowledge base covers the query terms and related concepts.", "code_blocks": [], "chunk_position": 3, "heading_path": "1. Coverage Score > 1. Coverage Score", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > 1. Coverage Score > 1. Coverage Score"}, {"id": "367eec5b85e296fb", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "Mathematical Foundation", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Mathematical Foundation\n\n```text\nCoverage(K, Q) = Σ(t ∈ Q) score(t, K) / |Q|\n\nwhere score(t, K) = doc_coverage(t) × (1 + freq_boost(t))\n```", "code_blocks": [{"language": "text", "code": "Coverage(K, Q) = Σ(t ∈ Q) score(t, K) / |Q|\n\nwhere score(t, K) = doc_coverage(t) × (1 + freq_boost(t))", "filename": ""}], "chunk_position": 3, "heading_path": "Mathematical Foundation > Mathematical Foundation", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > Mathematical Foundation > Mathematical Foundation"}, {"id": "a78ffaaa9d18954b", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "Components", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Components\n\n- **Document Coverage**: Percentage of documents containing the term\n- **Frequency Boost**: Logarithmic bonus for term frequency\n- **Query Decomposition**: Handles multi-word queries intelligently", "code_blocks": [], "chunk_position": 3, "heading_path": "Components > Components", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > Components > Components"}, {"id": "93c9d83b0bb97332", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "Tuning Coverage", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Tuning Coverage\n\n```python\n# For technical documentation with specific terminology\nconfig = AdaptiveConfig(\n    confidence_threshold=0.85,  # Require high coverage\n    top_k_links=5              # Cast wider net\n)\n\n# For…\n```", "code_blocks": [{"language": "python", "code": "# For technical documentation with specific terminology\nconfig = AdaptiveConfig(\n    confidence_threshold=0.85,  # Require high coverage\n    top_k_links=5              # Cast wider net\n)\n\n# For…", "filename": ""}], "chunk_position": 3, "heading_path": "Tuning Coverage > Tuning Coverage", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > Tuning Coverage > Tuning Coverage"}, {"id": "4079a84c98c4edca", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "2. Consistency Score", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: 2. Consistency Score\n\nConsistency evaluates whether the information across pages is coherent and non-contradictory.", "code_blocks": [], "chunk_position": 3, "heading_path": "2. Consistency Score > 2. Consistency Score", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > 2. Consistency Score > 2. Consistency Score"}, {"id": "8ca1a35a5d4963dc", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "How It Works", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: How It Works\n\n- Extracts key statements from each document\n- Compares statements across documents\n- Measures agreement vs. contradiction\n- Returns normalized score (0-1)", "code_blocks": [], "chunk_position": 3, "heading_path": "How It Works > How It Works", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > How It Works > How It Works"}, {"id": "3f3dd450f2a90ac1", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "Practical Impact", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Practical Impact\n\n- **High consistency (>0.8)**: Information is reliable and coherent\n- **Medium consistency (0.5-0.8)**: Some variation, but generally aligned\n- **Low consistency (<0.5)**: Conflicting information,…", "code_blocks": [], "chunk_position": 3, "heading_path": "Practical Impact > Practical Impact", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > Practical Impact > Practical Impact"}, {"id": "527f4378ad2a1a8a", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "3. Saturation Score", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: 3. Saturation Score\n\nSaturation detects when new pages stop providing novel information.", "code_blocks": [], "chunk_position": 3, "heading_path": "3. Saturation Score > 3. Saturation Score", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > 3. Saturation Score > 3. Saturation Score"}, {"id": "e4b96e4527e184b4", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "Detection Algorithm", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Detection Algorithm\n\n```python\n# Tracks new unique terms per page\nnew_terms_page_1 = 50\nnew_terms_page_2 = 30  # 60% of first\nnew_terms_page_3 = 15  # 50% of second\nnew_terms_page_4 = 5   # 33% of third\n# Saturation detected:…\n```", "code_blocks": [{"language": "python", "code": "# Tracks new unique terms per page\nnew_terms_page_1 = 50\nnew_terms_page_2 = 30  # 60% of first\nnew_terms_page_3 = 15  # 50% of second\nnew_terms_page_4 = 5   # 33% of third\n# Saturation detected:…", "filename": ""}], "chunk_position": 3, "heading_path": "Detection Algorithm > Detection Algorithm", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > Detection Algorithm > Detection Algorithm"}, {"id": "d43392316ea5dd24", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "Configuration", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Configuration\n\n```python\nconfig = AdaptiveConfig(\n    min_gain_threshold=0.1  # Stop if <10% new information\n)\n```", "code_blocks": [{"language": "python", "code": "config = AdaptiveConfig(\n    min_gain_threshold=0.1  # Stop if <10% new information\n)", "filename": ""}], "chunk_position": 3, "heading_path": "Configuration > Configuration", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > Configuration > Configuration"}, {"id": "a7f341fcecc63111", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "1. Relevance Scoring", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: 1. Relevance Scoring\n\nUses BM25 algorithm on link preview text:\n\nFactors:\n- Term frequency in preview\n- Inverse document frequency\n- Preview length normalization", "code_blocks": [{"language": "text", "code": "relevance = BM25(link.preview_text, query)", "filename": ""}], "chunk_position": 3, "heading_path": "1. Relevance Scoring > 1. Relevance Scoring", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > 1. Relevance Scoring > 1. Relevance Scoring"}, {"id": "976dad5751e85121", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "2. Novelty Estimation", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: 2. Novelty Estimation\n\nMeasures how different the link appears from already-crawled content:\n\nPrevents crawling duplicate or highly similar pages.", "code_blocks": [{"language": "text", "code": "novelty = 1 - max_similarity(preview, knowledge_base)", "filename": ""}], "chunk_position": 3, "heading_path": "2. Novelty Estimation > 2. Novelty Estimation", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > 2. Novelty Estimation > 2. Novelty Estimation"}, {"id": "59267dcae6a01866", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "3. Authority Calculation", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: 3. Authority Calculation\n\nURL structure and domain analysis:\n\nFactors:\n- Domain reputation\n- URL depth (fewer slashes = higher authority)\n- Clean URL structure", "code_blocks": [{"language": "text", "code": "authority = f(domain_rank, url_depth, url_structure)", "filename": ""}], "chunk_position": 3, "heading_path": "3. Authority Calculation > 3. Authority Calculation", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > 3. Authority Calculation > 3. Authority Calculation"}, {"id": "e1342753fae95a96", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "Technical Documentation", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Technical Documentation\n\nRationale:\n- High threshold ensures comprehensive coverage\n- Lower gain threshold captures edge cases\n- Moderate link following for depth", "code_blocks": [{"language": "python", "code": "tech_doc_config = AdaptiveConfig(\n    confidence_threshold=0.85,\n    max_pages=30,\n    top_k_links=3,\n    min_gain_threshold=0.05  # Keep crawling for small gains\n)", "filename": ""}], "chunk_position": 3, "heading_path": "Technical Documentation > Technical Documentation", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > Technical Documentation > Technical Documentation"}, {"id": "27745b954c182c48", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "News & Articles", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: News & Articles\n\nRationale:\n- Lower threshold (articles often repeat information)\n- Higher gain threshold (avoid duplicate stories)\n- More links per page (explore different perspectives)", "code_blocks": [{"language": "python", "code": "news_config = AdaptiveConfig(\n    confidence_threshold=0.6,\n    max_pages=10,\n    top_k_links=5,\n    min_gain_threshold=0.15  # Stop quickly on repetition\n)", "filename": ""}], "chunk_position": 3, "heading_path": "News & Articles > News & Articles", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > News & Articles > News & Articles"}, {"id": "5b8c774a8f57d85b", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "E-commerce", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: E-commerce\n\nRationale:\n- Balanced threshold for product variations\n- Focused link following (avoid infinite products)\n- Standard gain threshold", "code_blocks": [{"language": "python", "code": "ecommerce_config = AdaptiveConfig(\n    confidence_threshold=0.7,\n    max_pages=20,\n    top_k_links=2,\n    min_gain_threshold=0.1\n)", "filename": ""}], "chunk_position": 3, "heading_path": "E-commerce > E-commerce", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > E-commerce > E-commerce"}, {"id": "944230ae72de6d18", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "Research & Academic", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Research & Academic\n\nRationale:\n- Very high threshold for completeness\n- Many pages allowed for thorough research\n- Very low gain threshold to capture references", "code_blocks": [{"language": "python", "code": "research_config = AdaptiveConfig(\n    confidence_threshold=0.9,\n    max_pages=50,\n    top_k_links=4,\n    min_gain_threshold=0.02  # Very low - capture citations\n)", "filename": ""}], "chunk_position": 3, "heading_path": "Research & Academic > Research & Academic", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > Research & Academic > Research & Academic"}, {"id": "cf59582ad5590ab4", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "Memory Management", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Memory Management\n\n```python\n# For large crawls, use streaming\nconfig = AdaptiveConfig(\n    max_pages=100,\n    save_state=True,\n    state_path=\"large_crawl.json\"\n)\n\n# Periodically clean state\nif len(state.knowledge_base) >…\n```", "code_blocks": [{"language": "python", "code": "# For large crawls, use streaming\nconfig = AdaptiveConfig(\n    max_pages=100,\n    save_state=True,\n    state_path=\"large_crawl.json\"\n)\n\n# Periodically clean state\nif len(state.knowledge_base) >…", "filename": ""}], "chunk_position": 3, "heading_path": "Memory Management > Memory Management", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > Memory Management > Memory Management"}, {"id": "84da0ee32b723b6c", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "Parallel Processing", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Parallel Processing\n\n```python\n# Use multiple start points\nstart_urls = [\n    \"https://docs.example.com/intro\",\n    \"https://docs.example.com/api\",\n    \"https://docs.example.com/guides\"\n]\n\n# Crawl in parallel\ntasks = […\n```", "code_blocks": [{"language": "python", "code": "# Use multiple start points\nstart_urls = [\n    \"https://docs.example.com/intro\",\n    \"https://docs.example.com/api\",\n    \"https://docs.example.com/guides\"\n]\n\n# Crawl in parallel\ntasks = […", "filename": ""}], "chunk_position": 3, "heading_path": "Parallel Processing > Parallel Processing", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > Parallel Processing > Parallel Processing"}, {"id": "d1a6f8ea9eaa4ca0", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "Enable Debugging", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Enable Debugging\n\n```python\nimport logging\n\nlogging.basicConfig(level=logging.DEBUG)\nadaptive = AdaptiveCrawler(crawler, config, verbose=True)\n```", "code_blocks": [{"language": "python", "code": "import logging\n\nlogging.basicConfig(level=logging.DEBUG)\nadaptive = AdaptiveCrawler(crawler, config, verbose=True)", "filename": ""}], "chunk_position": 3, "heading_path": "Enable Debugging > Enable Debugging", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > Enable Debugging > Enable Debugging"}, {"id": "075ee497ca63e9af", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "Analyze Crawl Patterns", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Analyze Crawl Patterns\n\n```python\n# After crawling\nstate = await adaptive.digest(start_url, query)\n\n# Analyze link selection\nprint(\"Link selection order:\")\nfor i, url in enumerate(state.crawl_order):\n    print(f\"{i+1}. {url}\")\n\n#…\n```", "code_blocks": [{"language": "python", "code": "# After crawling\nstate = await adaptive.digest(start_url, query)\n\n# Analyze link selection\nprint(\"Link selection order:\")\nfor i, url in enumerate(state.crawl_order):\n    print(f\"{i+1}. {url}\")\n\n#…", "filename": ""}], "chunk_position": 3, "heading_path": "Analyze Crawl Patterns > Analyze Crawl Patterns", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > Analyze Crawl Patterns > Analyze Crawl Patterns"}, {"id": "566c3adfafb16a30", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "Export for Analysis", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Export for Analysis\n\n```python\n# Export detailed metrics\nimport json\n\nmetrics = {\n    \"query\": query,\n    \"total_pages\": len(state.crawled_urls),\n    \"confidence\": adaptive.confidence,\n    \"coverage_stats\":…\n```", "code_blocks": [{"language": "python", "code": "# Export detailed metrics\nimport json\n\nmetrics = {\n    \"query\": query,\n    \"total_pages\": len(state.crawled_urls),\n    \"confidence\": adaptive.confidence,\n    \"coverage_stats\":…", "filename": ""}], "chunk_position": 3, "heading_path": "Export for Analysis > Export for Analysis", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > Export for Analysis > Export for Analysis"}, {"id": "d046f798adccba7f", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "Implementing a Custom Strategy", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Implementing a Custom Strategy\n\n```python\nfrom crawl4ai.adaptive_crawler import CrawlStrategy\n\nclass DomainSpecificStrategy(CrawlStrategy):\n    def calculate_coverage(self, state: CrawlState) -> float:\n        # Custom coverage calculation…\n```", "code_blocks": [{"language": "python", "code": "from crawl4ai.adaptive_crawler import CrawlStrategy\n\nclass DomainSpecificStrategy(CrawlStrategy):\n    def calculate_coverage(self, state: CrawlState) -> float:\n        # Custom coverage calculation…", "filename": ""}], "chunk_position": 3, "heading_path": "Implementing a Custom Strategy > Implementing a Custom Strategy", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > Implementing a Custom Strategy > Implementing a Custom Strategy"}, {"id": "5b5112a9b83c3d32", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "Combining Strategies", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Combining Strategies\n\n```python\nclass HybridStrategy(CrawlStrategy):\n    def __init__(self):\n        self.strategies = [\n            TechnicalDocStrategy(),\n            SemanticSimilarityStrategy(),…\n```", "code_blocks": [{"language": "python", "code": "class HybridStrategy(CrawlStrategy):\n    def __init__(self):\n        self.strategies = [\n            TechnicalDocStrategy(),\n            SemanticSimilarityStrategy(),…", "filename": ""}], "chunk_position": 3, "heading_path": "Combining Strategies > Combining Strategies", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > Combining Strategies > Combining Strategies"}, {"id": "4f58d1fe3002529e", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "1. Start Conservative", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: 1. Start Conservative\n\nBegin with default settings and adjust based on results:", "code_blocks": [{"language": "python", "code": "# Start with defaults\nresult = await adaptive.digest(url, query)\n\n# Analyze and adjust\nif adaptive.confidence < 0.7:\n    config.max_pages += 10\n    config.confidence_threshold -= 0.1", "filename": ""}], "chunk_position": 3, "heading_path": "1. Start Conservative > 1. Start Conservative", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > 1. Start Conservative > 1. Start Conservative"}, {"id": "2145fbb1edef5dad", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "2. Monitor Resource Usage", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: 2. Monitor Resource Usage\n\n```python\nimport psutil\n\n# Check memory before large crawls\nmemory_percent = psutil.virtual_memory().percent\nif memory_percent > 80:\n    config.max_pages = min(config.max_pages, 20)\n```", "code_blocks": [{"language": "python", "code": "import psutil\n\n# Check memory before large crawls\nmemory_percent = psutil.virtual_memory().percent\nif memory_percent > 80:\n    config.max_pages = min(config.max_pages, 20)", "filename": ""}], "chunk_position": 3, "heading_path": "2. Monitor Resource Usage > 2. Monitor Resource Usage", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > 2. Monitor Resource Usage > 2. Monitor Resource Usage"}, {"id": "35383d578c044470", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "3. Use Domain Knowledge", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: 3. Use Domain Knowledge\n\n```python\n# For API documentation\nif \"api\" in start_url:\n    config.top_k_links = 2  # APIs have clear structure\n\n# For blogs\nif \"blog\" in start_url:\n    config.min_gain_threshold = 0.2  # Avoid similar posts\n```", "code_blocks": [{"language": "python", "code": "# For API documentation\nif \"api\" in start_url:\n    config.top_k_links = 2  # APIs have clear structure\n\n# For blogs\nif \"blog\" in start_url:\n    config.min_gain_threshold = 0.2  # Avoid similar posts", "filename": ""}], "chunk_position": 3, "heading_path": "3. Use Domain Knowledge > 3. Use Domain Knowledge", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > 3. Use Domain Knowledge > 3. Use Domain Knowledge"}, {"id": "68bea90473714234", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "4. Validate Results", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: 4. Validate Results\n\n```python\n# Always validate the knowledge base\nrelevant_content = adaptive.get_relevant_content(top_k=10)\n\n# Check coverage\nquery_terms = set(query.lower().split())\ncovered_terms = set()\n\nfor doc in…\n```", "code_blocks": [{"language": "python", "code": "# Always validate the knowledge base\nrelevant_content = adaptive.get_relevant_content(top_k=10)\n\n# Check coverage\nquery_terms = set(query.lower().split())\ncovered_terms = set()\n\nfor doc in…", "filename": ""}], "chunk_position": 3, "heading_path": "4. Validate Results > 4. Validate Results", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > 4. Validate Results > 4. Validate Results"}, {"id": "35dd18f181ba0530", "url": "https://docs.crawl4ai.com/advanced/adaptive-strategies/", "page_title": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide covers advanced adaptive strategies in Crawl4AI, including the three-layer scoring system, link ranking algorithm, domain-specific configurations, performance optimization, debugging,…", "heading": "Next Steps", "content": "Page: Adaptive Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Next Steps\n\nExplore [Custom Strategy Implementation](../tutorials/custom-adaptive-strategies.md)\nLearn about [Knowledge Base Management](../tutorials/knowledge-base-management.md)\nSee [Performance…", "code_blocks": [], "chunk_position": 3, "heading_path": "Next Steps > Next Steps", "breadcrumbs": "Adaptive Strategies - Crawl4AI Documentation (v0.9.x) > Next Steps > Next Steps"}, {"id": "36603dcee6bfb301", "url": "https://docs.crawl4ai.com/advanced/advanced-features/", "page_title": "Overview of Some Important Advanced Features", "page_type": "guide", "page_summary": "A guide to advanced Crawl4AI features including proxy usage, PDF/screenshot capture, SSL certificates, custom headers, session persistence, robots.txt compliance, and anti-bot techniques.", "heading": "Overview of Some Important Advanced Features", "content": "Page: Overview of Some Important Advanced Features\nSection: Overview of Some Important Advanced Features\n\nCrawl4AI offers multiple power-user features that go beyond simple crawling. This tutorial covers:\n\n1. **Proxy Usage**\n\n2. **Capturing PDFs & Screenshots**\n\n3. **Handling SSL Certificates**\n\n4.…", "code_blocks": [], "chunk_position": 4, "heading_path": "Overview of Some Important Advanced Features > Overview of Some Important Advanced Features", "breadcrumbs": "Overview of Some Important Advanced Features > Overview of Some Important Advanced Features > Overview of Some Important Advanced Features"}, {"id": "35081abc0cec41f2", "url": "https://docs.crawl4ai.com/advanced/advanced-features/", "page_title": "Overview of Some Important Advanced Features", "page_type": "guide", "page_summary": "A guide to advanced Crawl4AI features including proxy usage, PDF/screenshot capture, SSL certificates, custom headers, session persistence, robots.txt compliance, and anti-bot techniques.", "heading": "1. Proxy Usage", "content": "Page: Overview of Some Important Advanced Features\nSection: 1. Proxy Usage\n\nIf you need to route your crawl traffic through a proxy—whether for IP rotation, geo-testing, or privacy—Crawl4AI supports it via `BrowserConfig.proxy_config`.\n\n**Key Points**\n\n- **`proxy_config`**…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig\n\nasync def main():\n    browser_cfg = BrowserConfig(\n        proxy_config={\n            \"server\":…", "filename": ""}], "chunk_position": 4, "heading_path": "1. Proxy Usage > 1. Proxy Usage", "breadcrumbs": "Overview of Some Important Advanced Features > 1. Proxy Usage > 1. Proxy Usage"}, {"id": "8ffb9469e044b449", "url": "https://docs.crawl4ai.com/advanced/advanced-features/", "page_title": "Overview of Some Important Advanced Features", "page_type": "guide", "page_summary": "A guide to advanced Crawl4AI features including proxy usage, PDF/screenshot capture, SSL certificates, custom headers, session persistence, robots.txt compliance, and anti-bot techniques.", "heading": "2. Capturing PDFs & Screenshots", "content": "Page: Overview of Some Important Advanced Features\nSection: 2. Capturing PDFs & Screenshots\n\nSometimes you need a visual record of a page or a PDF “printout.” Crawl4AI can do both in one pass:\n\n**Why PDF + Screenshot?**\n\n- Large or complex pages can be slow or error-prone with “traditional”…", "code_blocks": [{"language": "python", "code": "import os, asyncio\nfrom base64 import b64decode\nfrom crawl4ai import AsyncWebCrawler, CacheMode, CrawlerRunConfig\n\nasync def main():\n    run_config = CrawlerRunConfig(…", "filename": ""}], "chunk_position": 4, "heading_path": "2. Capturing PDFs & Screenshots > 2. Capturing PDFs & Screenshots", "breadcrumbs": "Overview of Some Important Advanced Features > 2. Capturing PDFs & Screenshots > 2. Capturing PDFs & Screenshots"}, {"id": "ae1f0c818d08c71f", "url": "https://docs.crawl4ai.com/advanced/advanced-features/", "page_title": "Overview of Some Important Advanced Features", "page_type": "guide", "page_summary": "A guide to advanced Crawl4AI features including proxy usage, PDF/screenshot capture, SSL certificates, custom headers, session persistence, robots.txt compliance, and anti-bot techniques.", "heading": "3. Handling SSL Certificates", "content": "Page: Overview of Some Important Advanced Features\nSection: 3. Handling SSL Certificates\n\nIf you need to verify or export a site’s SSL certificate—for compliance, debugging, or data analysis—Crawl4AI can fetch it during the crawl:\n\n**Key Points**\n\n- **`fetch_ssl_certificate=True`**…", "code_blocks": [{"language": "python", "code": "import asyncio, os\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\n\nasync def main():\n    tmp_dir = os.path.join(os.getcwd(), \"tmp\")\n    os.makedirs(tmp_dir, exist_ok=True)…", "filename": ""}], "chunk_position": 4, "heading_path": "3. Handling SSL Certificates > 3. Handling SSL Certificates", "breadcrumbs": "Overview of Some Important Advanced Features > 3. Handling SSL Certificates > 3. Handling SSL Certificates"}, {"id": "f0a11cd28d68fd77", "url": "https://docs.crawl4ai.com/advanced/advanced-features/", "page_title": "Overview of Some Important Advanced Features", "page_type": "guide", "page_summary": "A guide to advanced Crawl4AI features including proxy usage, PDF/screenshot capture, SSL certificates, custom headers, session persistence, robots.txt compliance, and anti-bot techniques.", "heading": "4. Custom Headers", "content": "Page: Overview of Some Important Advanced Features\nSection: 4. Custom Headers\n\nSometimes you need to set custom headers (e.g., language preferences, authentication tokens, or specialized user-agent strings). You can do this in multiple ways:\n\n**Notes**\n\n- Some sites may react…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler\n\nasync def main():\n    # Option 1: Set headers at the crawler strategy level\n    crawler1 = AsyncWebCrawler(\n        # The underlying strategy can…", "filename": ""}], "chunk_position": 4, "heading_path": "4. Custom Headers > 4. Custom Headers", "breadcrumbs": "Overview of Some Important Advanced Features > 4. Custom Headers > 4. Custom Headers"}, {"id": "24f902ee56550160", "url": "https://docs.crawl4ai.com/advanced/advanced-features/", "page_title": "Overview of Some Important Advanced Features", "page_type": "guide", "page_summary": "A guide to advanced Crawl4AI features including proxy usage, PDF/screenshot capture, SSL certificates, custom headers, session persistence, robots.txt compliance, and anti-bot techniques.", "heading": "5. Session Persistence & Local Storage", "content": "Page: Overview of Some Important Advanced Features\nSection: 5. Session Persistence & Local Storage\n\nCrawl4AI can preserve cookies and localStorage so you can continue where you left off—ideal for logging into sites or skipping repeated auth flows.", "code_blocks": [], "chunk_position": 4, "heading_path": "5. Session Persistence & Local Storage > 5. Session Persistence & Local Storage", "breadcrumbs": "Overview of Some Important Advanced Features > 5. Session Persistence & Local Storage > 5. Session Persistence & Local Storage"}, {"id": "836f9e5dc6e083fb", "url": "https://docs.crawl4ai.com/advanced/advanced-features/", "page_title": "Overview of Some Important Advanced Features", "page_type": "guide", "page_summary": "A guide to advanced Crawl4AI features including proxy usage, PDF/screenshot capture, SSL certificates, custom headers, session persistence, robots.txt compliance, and anti-bot techniques.", "heading": "5.1 `storage_state`", "content": "Page: Overview of Some Important Advanced Features\nSection: 5.1 `storage_state`\n\n```python\nimport asyncio\nfrom crawl4ai import AsyncWebCrawler\n\nasync def main():\n    storage_dict = {\n        \"cookies\": [\n            {\n                \"name\": \"session\",\n                \"value\": \"abcd1234\",…\n```", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler\n\nasync def main():\n    storage_dict = {\n        \"cookies\": [\n            {\n                \"name\": \"session\",\n                \"value\": \"abcd1234\",…", "filename": ""}], "chunk_position": 4, "heading_path": "5.1 `storage_state` > 5.1 `storage_state`", "breadcrumbs": "Overview of Some Important Advanced Features > 5.1 `storage_state` > 5.1 `storage_state`"}, {"id": "b723a2e7c9be1a27", "url": "https://docs.crawl4ai.com/advanced/advanced-features/", "page_title": "Overview of Some Important Advanced Features", "page_type": "guide", "page_summary": "A guide to advanced Crawl4AI features including proxy usage, PDF/screenshot capture, SSL certificates, custom headers, session persistence, robots.txt compliance, and anti-bot techniques.", "heading": "5.2 Exporting & Reusing State", "content": "Page: Overview of Some Important Advanced Features\nSection: 5.2 Exporting & Reusing State\n\nYou can sign in once, export the browser context, and reuse it later—without re-entering credentials.\n\n- **`await context.storage_state(path=\"my_storage.json\")`**: Exports cookies, localStorage, etc.…", "code_blocks": [], "chunk_position": 4, "heading_path": "5.2 Exporting & Reusing State > 5.2 Exporting & Reusing State", "breadcrumbs": "Overview of Some Important Advanced Features > 5.2 Exporting & Reusing State > 5.2 Exporting & Reusing State"}, {"id": "b75df52ddd327dbe", "url": "https://docs.crawl4ai.com/advanced/advanced-features/", "page_title": "Overview of Some Important Advanced Features", "page_type": "guide", "page_summary": "A guide to advanced Crawl4AI features including proxy usage, PDF/screenshot capture, SSL certificates, custom headers, session persistence, robots.txt compliance, and anti-bot techniques.", "heading": "6. Robots.txt Compliance", "content": "Page: Overview of Some Important Advanced Features\nSection: 6. Robots.txt Compliance\n\nCrawl4AI supports respecting robots.txt rules with efficient caching:\n\n**Key Points**\n\n- Robots.txt files are cached locally for efficiency\n\n- Cache is stored in…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nasync def main():\n    # Enable robots.txt checking in config\n    config = CrawlerRunConfig(\n        check_robots_txt=True  #…", "filename": ""}], "chunk_position": 4, "heading_path": "6. Robots.txt Compliance > 6. Robots.txt Compliance", "breadcrumbs": "Overview of Some Important Advanced Features > 6. Robots.txt Compliance > 6. Robots.txt Compliance"}, {"id": "de02a5a0721def27", "url": "https://docs.crawl4ai.com/advanced/advanced-features/", "page_title": "Overview of Some Important Advanced Features", "page_type": "guide", "page_summary": "A guide to advanced Crawl4AI features including proxy usage, PDF/screenshot capture, SSL certificates, custom headers, session persistence, robots.txt compliance, and anti-bot techniques.", "heading": "Putting It All Together", "content": "Page: Overview of Some Important Advanced Features\nSection: Putting It All Together\n\nHere’s a snippet that combines multiple “advanced” features (proxy, PDF, screenshot, SSL, custom headers, and session reuse) into one run. Normally, you’d tailor each setting to your project’s needs.", "code_blocks": [{"language": "python", "code": "import os, asyncio\nfrom base64 import b64decode\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode\n\nasync def main():\n    # 1. Browser config with proxy + headless…", "filename": ""}], "chunk_position": 4, "heading_path": "Putting It All Together > Putting It All Together", "breadcrumbs": "Overview of Some Important Advanced Features > Putting It All Together > Putting It All Together"}, {"id": "93adb066cc850beb", "url": "https://docs.crawl4ai.com/advanced/advanced-features/", "page_title": "Overview of Some Important Advanced Features", "page_type": "guide", "page_summary": "A guide to advanced Crawl4AI features including proxy usage, PDF/screenshot capture, SSL certificates, custom headers, session persistence, robots.txt compliance, and anti-bot techniques.", "heading": "7. Anti-Bot Features (Stealth Mode & Undetected Browser)", "content": "Page: Overview of Some Important Advanced Features\nSection: 7. Anti-Bot Features (Stealth Mode & Undetected Browser)\n\nCrawl4AI provides two powerful features to bypass bot detection:", "code_blocks": [], "chunk_position": 4, "heading_path": "7. Anti-Bot Features (Stealth Mode & Undetected Browser) > 7. Anti-Bot Features (Stealth Mode & Undetected Browser)", "breadcrumbs": "Overview of Some Important Advanced Features > 7. Anti-Bot Features (Stealth Mode & Undetected Browser) > 7. Anti-Bot Features (Stealth Mode & Undetected Browser)"}, {"id": "e78c8340fd53f20a", "url": "https://docs.crawl4ai.com/advanced/advanced-features/", "page_title": "Overview of Some Important Advanced Features", "page_type": "guide", "page_summary": "A guide to advanced Crawl4AI features including proxy usage, PDF/screenshot capture, SSL certificates, custom headers, session persistence, robots.txt compliance, and anti-bot techniques.", "heading": "7.1 Stealth Mode", "content": "Page: Overview of Some Important Advanced Features\nSection: 7.1 Stealth Mode\n\nStealth mode uses playwright-stealth to modify browser fingerprints and behaviors. Enable it with a simple flag:\n\n**When to use**: Sites with basic bot detection (checking navigator.webdriver,…", "code_blocks": [{"language": "python", "code": "browser_config = BrowserConfig(\n    enable_stealth=True,  # Activates stealth mode\n    headless=False\n)", "filename": ""}], "chunk_position": 4, "heading_path": "7.1 Stealth Mode > 7.1 Stealth Mode", "breadcrumbs": "Overview of Some Important Advanced Features > 7.1 Stealth Mode > 7.1 Stealth Mode"}, {"id": "5e9ada3dc8acd3df", "url": "https://docs.crawl4ai.com/advanced/advanced-features/", "page_title": "Overview of Some Important Advanced Features", "page_type": "guide", "page_summary": "A guide to advanced Crawl4AI features including proxy usage, PDF/screenshot capture, SSL certificates, custom headers, session persistence, robots.txt compliance, and anti-bot techniques.", "heading": "7.2 Undetected Browser", "content": "Page: Overview of Some Important Advanced Features\nSection: 7.2 Undetected Browser\n\nFor advanced bot detection, use the undetected browser adapter:\n\n**When to use**: Sites with sophisticated bot detection (Cloudflare, DataDome, etc.)", "code_blocks": [{"language": "python", "code": "from crawl4ai import UndetectedAdapter\nfrom crawl4ai.async_crawler_strategy import AsyncPlaywrightCrawlerStrategy\n\n# Create undetected adapter\nadapter = UndetectedAdapter()\nstrategy =…", "filename": ""}], "chunk_position": 4, "heading_path": "7.2 Undetected Browser > 7.2 Undetected Browser", "breadcrumbs": "Overview of Some Important Advanced Features > 7.2 Undetected Browser > 7.2 Undetected Browser"}, {"id": "1601e9f5adea1f82", "url": "https://docs.crawl4ai.com/advanced/advanced-features/", "page_title": "Overview of Some Important Advanced Features", "page_type": "guide", "page_summary": "A guide to advanced Crawl4AI features including proxy usage, PDF/screenshot capture, SSL certificates, custom headers, session persistence, robots.txt compliance, and anti-bot techniques.", "heading": "7.3 Combining Both", "content": "Page: Overview of Some Important Advanced Features\nSection: 7.3 Combining Both\n\nFor maximum evasion, combine stealth mode with undetected browser:", "code_blocks": [{"language": "python", "code": "browser_config = BrowserConfig(\n    enable_stealth=True,  # Enable stealth\n    headless=False\n)\n\nadapter = UndetectedAdapter()  # Use undetected browser", "filename": ""}], "chunk_position": 4, "heading_path": "7.3 Combining Both > 7.3 Combining Both", "breadcrumbs": "Overview of Some Important Advanced Features > 7.3 Combining Both > 7.3 Combining Both"}, {"id": "2a99440c477fe2e7", "url": "https://docs.crawl4ai.com/advanced/advanced-features/", "page_title": "Overview of Some Important Advanced Features", "page_type": "guide", "page_summary": "A guide to advanced Crawl4AI features including proxy usage, PDF/screenshot capture, SSL certificates, custom headers, session persistence, robots.txt compliance, and anti-bot techniques.", "heading": "Choosing the Right Approach", "content": "Page: Overview of Some Important Advanced Features\nSection: Choosing the Right Approach\n\n| Detection Level | Recommended Approach |\n| --- | --- |\n| No protection | Regular browser |\n| Basic checks | Regular + Stealth mode |\n| Advanced protection | Undetected browser |\n| Maximum evasion |…", "code_blocks": [], "chunk_position": 4, "heading_path": "Choosing the Right Approach > Choosing the Right Approach", "breadcrumbs": "Overview of Some Important Advanced Features > Choosing the Right Approach > Choosing the Right Approach"}, {"id": "81c89104f8e4ede4", "url": "https://docs.crawl4ai.com/advanced/advanced-features/", "page_title": "Overview of Some Important Advanced Features", "page_type": "guide", "page_summary": "A guide to advanced Crawl4AI features including proxy usage, PDF/screenshot capture, SSL certificates, custom headers, session persistence, robots.txt compliance, and anti-bot techniques.", "heading": "Conclusion & Next Steps", "content": "Page: Overview of Some Important Advanced Features\nSection: Conclusion & Next Steps\n\nYou've now explored several **advanced** features:\n\n- **Proxy Usage**\n\n- **PDF & Screenshot** capturing for large or critical pages\n\n- **SSL Certificate** retrieval & exporting\n\n- **Custom Headers**…", "code_blocks": [], "chunk_position": 4, "heading_path": "Conclusion & Next Steps > Conclusion & Next Steps", "breadcrumbs": "Overview of Some Important Advanced Features > Conclusion & Next Steps > Conclusion & Next Steps"}, {"id": "4a648cb9b274678c", "url": "https://docs.crawl4ai.com/advanced/anti-bot-and-fallback/", "page_title": "Anti-Bot Detection & Fallback", "page_type": "guide", "page_summary": "Explains how Crawl4AI detects anti-bot blocks and describes the layered retry/fallback system using proxies, retries, and custom fetch functions to retrieve content from protected sites.", "heading": "How Detection Works", "content": "Page: Anti-Bot Detection & Fallback\nSection: How Detection Works\n\nAfter each crawl attempt, Crawl4AI inspects the HTTP status code and HTML content for known anti-bot signals:\n\n- **HTTP 403/429** with short or empty response bodies\n- **Challenge pages** —…", "code_blocks": [], "chunk_position": 5, "heading_path": "How Detection Works > How Detection Works", "breadcrumbs": "Anti-Bot Detection & Fallback > How Detection Works > How Detection Works"}, {"id": "63406b3f0ed40584", "url": "https://docs.crawl4ai.com/advanced/anti-bot-and-fallback/", "page_title": "Anti-Bot Detection & Fallback", "page_type": "guide", "page_summary": "Explains how Crawl4AI detects anti-bot blocks and describes the layered retry/fallback system using proxies, retries, and custom fetch functions to retrieve content from protected sites.", "heading": "Configuration Options", "content": "Page: Anti-Bot Detection & Fallback\nSection: Configuration Options\n\nAll anti-bot retry options live on `CrawlerRunConfig`:\n\n| Parameter | Type | Default | Description |\n| --- | --- | --- | --- |\n| `proxy_config` | `ProxyConfig`, `list[ProxyConfig]`, or `None` |…", "code_blocks": [], "chunk_position": 5, "heading_path": "Configuration Options > Configuration Options", "breadcrumbs": "Anti-Bot Detection & Fallback > Configuration Options > Configuration Options"}, {"id": "e0c1d955aa8c59f4", "url": "https://docs.crawl4ai.com/advanced/anti-bot-and-fallback/", "page_title": "Anti-Bot Detection & Fallback", "page_type": "guide", "page_summary": "Explains how Crawl4AI detects anti-bot blocks and describes the layered retry/fallback system using proxies, retries, and custom fetch functions to retrieve content from protected sites.", "heading": "Escalation Chain", "content": "Page: Anti-Bot Detection & Fallback\nSection: Escalation Chain\n\nEach retry round tries every proxy in `proxy_config` in order. If all rounds are exhausted and the page is still blocked, the fallback fetch function is called as a last resort.\n\nWorst-case attempts…", "code_blocks": [{"language": "text", "code": "For each round (1 + max_retries rounds):\n    1. Try proxy_config[0] (or direct if proxy_config is None)\n    2. If blocked → try proxy_config[1]\n    3. If blocked → try proxy_config[2]\n    4. ...…", "filename": ""}], "chunk_position": 5, "heading_path": "Escalation Chain > Escalation Chain", "breadcrumbs": "Anti-Bot Detection & Fallback > Escalation Chain > Escalation Chain"}, {"id": "5f8cd717f8607c2c", "url": "https://docs.crawl4ai.com/advanced/anti-bot-and-fallback/", "page_title": "Anti-Bot Detection & Fallback", "page_type": "guide", "page_summary": "Explains how Crawl4AI detects anti-bot blocks and describes the layered retry/fallback system using proxies, retries, and custom fetch functions to retrieve content from protected sites.", "heading": "Crawl Stats", "content": "Page: Anti-Bot Detection & Fallback\nSection: Crawl Stats\n\nEvery crawl result includes a `crawl_stats` dict with detailed attempt tracking:", "code_blocks": [{"language": "python", "code": "result.crawl_stats = {\n    \"attempts\": 3,                    # total browser attempts made\n    \"retries\": 1,                     # retry rounds used (0 = succeeded first round)\n    \"proxies_used\": […", "filename": ""}], "chunk_position": 5, "heading_path": "Crawl Stats > Crawl Stats", "breadcrumbs": "Anti-Bot Detection & Fallback > Crawl Stats > Crawl Stats"}, {"id": "2be1ae0a815003c5", "url": "https://docs.crawl4ai.com/advanced/anti-bot-and-fallback/", "page_title": "Anti-Bot Detection & Fallback", "page_type": "guide", "page_summary": "Explains how Crawl4AI detects anti-bot blocks and describes the layered retry/fallback system using proxies, retries, and custom fetch functions to retrieve content from protected sites.", "heading": "Simple Retry (No Proxy)", "content": "Page: Anti-Bot Detection & Fallback\nSection: Simple Retry (No Proxy)\n\nRetry the crawl up to 3 times when blocking is detected. Useful when blocks are intermittent or IP-based.", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler\nfrom crawl4ai.async_configs import BrowserConfig, CrawlerRunConfig\n\nasync with AsyncWebCrawler(config=BrowserConfig(headless=True)) as crawler:\n    result = await…", "filename": ""}], "chunk_position": 5, "heading_path": "Simple Retry (No Proxy) > Simple Retry (No Proxy)", "breadcrumbs": "Anti-Bot Detection & Fallback > Simple Retry (No Proxy) > Simple Retry (No Proxy)"}, {"id": "4611cef3e69cb03f", "url": "https://docs.crawl4ai.com/advanced/anti-bot-and-fallback/", "page_title": "Anti-Bot Detection & Fallback", "page_type": "guide", "page_summary": "Explains how Crawl4AI detects anti-bot blocks and describes the layered retry/fallback system using proxies, retries, and custom fetch functions to retrieve content from protected sites.", "heading": "Single Proxy", "content": "Page: Anti-Bot Detection & Fallback\nSection: Single Proxy\n\nPass a single `ProxyConfig` — it's used on every attempt. Same behavior as always.", "code_blocks": [{"language": "python", "code": "from crawl4ai.async_configs import ProxyConfig\n\nconfig = CrawlerRunConfig(\n    max_retries=2,\n    proxy_config=ProxyConfig(\n        server=\"http://proxy.example.com:8080\",\n        username=\"user\",…", "filename": ""}], "chunk_position": 5, "heading_path": "Single Proxy > Single Proxy", "breadcrumbs": "Anti-Bot Detection & Fallback > Single Proxy > Single Proxy"}, {"id": "2535157d9aee0488", "url": "https://docs.crawl4ai.com/advanced/anti-bot-and-fallback/", "page_title": "Anti-Bot Detection & Fallback", "page_type": "guide", "page_summary": "Explains how Crawl4AI detects anti-bot blocks and describes the layered retry/fallback system using proxies, retries, and custom fetch functions to retrieve content from protected sites.", "heading": "Direct-First, Then Proxies", "content": "Page: Anti-Bot Detection & Fallback\nSection: Direct-First, Then Proxies\n\nTry without a proxy first, then escalate to proxies if blocked. Use `ProxyConfig.DIRECT` (or the string `\"direct\"`) in the list to represent a no-proxy attempt.\n\nWith this setup, each round tries…", "code_blocks": [{"language": "python", "code": "config = CrawlerRunConfig(\n    max_retries=1,\n    proxy_config=[\n        ProxyConfig.DIRECT,  # Try without proxy first\n        ProxyConfig(…", "filename": ""}], "chunk_position": 5, "heading_path": "Direct-First, Then Proxies > Direct-First, Then Proxies", "breadcrumbs": "Anti-Bot Detection & Fallback > Direct-First, Then Proxies > Direct-First, Then Proxies"}, {"id": "c3483e904dbd07c7", "url": "https://docs.crawl4ai.com/advanced/anti-bot-and-fallback/", "page_title": "Anti-Bot Detection & Fallback", "page_type": "guide", "page_summary": "Explains how Crawl4AI detects anti-bot blocks and describes the layered retry/fallback system using proxies, retries, and custom fetch functions to retrieve content from protected sites.", "heading": "Proxy List (Escalation)", "content": "Page: Anti-Bot Detection & Fallback\nSection: Proxy List (Escalation)\n\nPass a list of proxies. They're tried in order — first one that works wins. Within each retry round, the entire list is tried again.\n\nWith this setup, each round tries the datacenter proxy first,…", "code_blocks": [{"language": "python", "code": "config = CrawlerRunConfig(\n    max_retries=1,\n    proxy_config=[\n        ProxyConfig(\n            server=\"http://datacenter-proxy.example.com:8080\",\n            username=\"user\",…", "filename": ""}], "chunk_position": 5, "heading_path": "Proxy List (Escalation) > Proxy List (Escalation)", "breadcrumbs": "Anti-Bot Detection & Fallback > Proxy List (Escalation) > Proxy List (Escalation)"}, {"id": "203e99a4529a401c", "url": "https://docs.crawl4ai.com/advanced/anti-bot-and-fallback/", "page_title": "Anti-Bot Detection & Fallback", "page_type": "guide", "page_summary": "Explains how Crawl4AI detects anti-bot blocks and describes the layered retry/fallback system using proxies, retries, and custom fetch functions to retrieve content from protected sites.", "heading": "Fallback Fetch Function", "content": "Page: Anti-Bot Detection & Fallback\nSection: Fallback Fetch Function\n\nWhen all browser-based attempts fail, call a custom async function as a last resort. This function receives the URL and must return raw HTML as a string. The returned HTML is processed through the…", "code_blocks": [{"language": "python", "code": "import aiohttp\n\nasync def my_scraping_api(url: str) -> str:\n    \"\"\"Fetch HTML via an external scraping API.\"\"\"\n    async with aiohttp.ClientSession() as session:\n        async with session.get(…", "filename": ""}], "chunk_position": 5, "heading_path": "Fallback Fetch Function > Fallback Fetch Function", "breadcrumbs": "Anti-Bot Detection & Fallback > Fallback Fetch Function > Fallback Fetch Function"}, {"id": "bb79130a77b1c6a2", "url": "https://docs.crawl4ai.com/advanced/anti-bot-and-fallback/", "page_title": "Anti-Bot Detection & Fallback", "page_type": "guide", "page_summary": "Explains how Crawl4AI detects anti-bot blocks and describes the layered retry/fallback system using proxies, retries, and custom fetch functions to retrieve content from protected sites.", "heading": "Full Escalation (All Features Combined)", "content": "Page: Anti-Bot Detection & Fallback\nSection: Full Escalation (All Features Combined)\n\nThis example combines every layer: stealth mode, a list of proxies tried in order, retries, and a final fetch function.\n\n**What happens step by step:**\n\n| Round | Attempt | What runs |\n| --- | --- |…", "code_blocks": [{"language": "python", "code": "import aiohttp\nfrom crawl4ai import AsyncWebCrawler\nfrom crawl4ai.async_configs import BrowserConfig, CrawlerRunConfig, ProxyConfig\n\n# Last-resort: fetch HTML via an external service\nasync def…", "filename": ""}], "chunk_position": 5, "heading_path": "Full Escalation (All Features Combined) > Full Escalation (All Features Combined)", "breadcrumbs": "Anti-Bot Detection & Fallback > Full Escalation (All Features Combined) > Full Escalation (All Features Combined)"}, {"id": "61e14eacae70cfb8", "url": "https://docs.crawl4ai.com/advanced/anti-bot-and-fallback/", "page_title": "Anti-Bot Detection & Fallback", "page_type": "guide", "page_summary": "Explains how Crawl4AI detects anti-bot blocks and describes the layered retry/fallback system using proxies, retries, and custom fetch functions to retrieve content from protected sites.", "heading": "Tips", "content": "Page: Anti-Bot Detection & Fallback\nSection: Tips\n\n- Start with `max_retries=0` and a `fallback_fetch_function` if you just want a safety net without burning time on retries.\n- Order proxies cheapest-first — datacenter proxies before residential,…", "code_blocks": [], "chunk_position": 5, "heading_path": "Tips > Tips", "breadcrumbs": "Anti-Bot Detection & Fallback > Tips > Tips"}, {"id": "5ef30b9744607220", "url": "https://docs.crawl4ai.com/advanced/anti-bot-and-fallback/", "page_title": "Anti-Bot Detection & Fallback", "page_type": "guide", "page_summary": "Explains how Crawl4AI detects anti-bot blocks and describes the layered retry/fallback system using proxies, retries, and custom fetch functions to retrieve content from protected sites.", "heading": "See Also", "content": "Page: Anti-Bot Detection & Fallback\nSection: See Also\n\n- [Proxy & Security](../proxy-security/) — Proxy setup, authentication, and rotation\n- [Undetected Browser](../undetected-browser/) — Stealth mode and browser fingerprint evasion\n- [Session…", "code_blocks": [], "chunk_position": 5, "heading_path": "See Also > See Also", "breadcrumbs": "Anti-Bot Detection & Fallback > See Also > See Also"}, {"id": "6351e2cc37a4fc36", "url": "https://docs.crawl4ai.com/advanced/crawl-dispatcher/", "page_title": "Crawl Dispatcher", "page_type": "overview", "page_summary": "This page announces the upcoming Crawl Dispatcher module in Crawl4AI, a feature for handling thousands of crawling tasks simultaneously with efficient resource management and real-time monitoring.", "heading": "Crawl Dispatcher", "content": "Page: Crawl Dispatcher\nSection: Crawl Dispatcher\n\nWe’re excited to announce a  **Crawl Dispatcher**  module that can handle  **thousands**  of crawling tasks simultaneously. By efficiently managing system resources (memory, CPU, network), this…", "code_blocks": [], "chunk_position": 6, "heading_path": "Crawl Dispatcher > Crawl Dispatcher", "breadcrumbs": "Crawl Dispatcher > Crawl Dispatcher > Crawl Dispatcher"}, {"id": "84f66edf20b59892", "url": "https://docs.crawl4ai.com/advanced/file-downloading/", "page_title": "Download Handling in Crawl4AI", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI to handle file downloads during crawling, including enabling downloads, specifying download locations, triggering downloads, and accessing downloaded files.", "heading": "Overview", "content": "Page: Download Handling in Crawl4AI\nSection: Overview\n\nThis guide explains how to use Crawl4AI to handle file downloads during crawling. You'll learn how to trigger downloads, specify download locations, and access downloaded files.", "code_blocks": [], "chunk_position": 7, "heading_path": "Overview > Overview", "breadcrumbs": "Download Handling in Crawl4AI > Overview > Overview"}, {"id": "8aaf9bec5f78181c", "url": "https://docs.crawl4ai.com/advanced/file-downloading/", "page_title": "Download Handling in Crawl4AI", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI to handle file downloads during crawling, including enabling downloads, specifying download locations, triggering downloads, and accessing downloaded files.", "heading": "Enabling Downloads", "content": "Page: Download Handling in Crawl4AI\nSection: Enabling Downloads\n\nTo enable downloads, set the `accept_downloads` parameter in the `BrowserConfig` object and pass it to the crawler.", "code_blocks": [{"language": "", "code": "from crawl4ai.async_configs import BrowserConfig, AsyncWebCrawler\n\nasync def main():\n    config = BrowserConfig(accept_downloads=True)  # Enable downloads globally\n    async with…", "filename": ""}], "chunk_position": 7, "heading_path": "Enabling Downloads > Enabling Downloads", "breadcrumbs": "Download Handling in Crawl4AI > Enabling Downloads > Enabling Downloads"}, {"id": "70f5abba4dd8bf96", "url": "https://docs.crawl4ai.com/advanced/file-downloading/", "page_title": "Download Handling in Crawl4AI", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI to handle file downloads during crawling, including enabling downloads, specifying download locations, triggering downloads, and accessing downloaded files.", "heading": "Specifying Download Location", "content": "Page: Download Handling in Crawl4AI\nSection: Specifying Download Location\n\nSpecify the download directory using the `downloads_path` attribute in the `BrowserConfig` object. If not provided, Crawl4AI defaults to creating a \"downloads\" directory inside the `.crawl4ai` folder…", "code_blocks": [{"language": "", "code": "from crawl4ai.async_configs import BrowserConfig\nimport os\n\ndownloads_path = os.path.join(os.getcwd(), \"my_downloads\")  # Custom download path\nos.makedirs(downloads_path, exist_ok=True)\n\nconfig =…", "filename": ""}], "chunk_position": 7, "heading_path": "Specifying Download Location > Specifying Download Location", "breadcrumbs": "Download Handling in Crawl4AI > Specifying Download Location > Specifying Download Location"}, {"id": "d024df8a746ec412", "url": "https://docs.crawl4ai.com/advanced/file-downloading/", "page_title": "Download Handling in Crawl4AI", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI to handle file downloads during crawling, including enabling downloads, specifying download locations, triggering downloads, and accessing downloaded files.", "heading": "Triggering Downloads", "content": "Page: Download Handling in Crawl4AI\nSection: Triggering Downloads\n\nDownloads are typically triggered by user interactions on a web page, such as clicking a download button. Use `js_code` in `CrawlerRunConfig` to simulate these actions and `wait_for` to allow…", "code_blocks": [{"language": "", "code": "from crawl4ai.async_configs import CrawlerRunConfig\n\nconfig = CrawlerRunConfig(\n    js_code=\"\"\"\n        const downloadLink = document.querySelector('a[href$=\".exe\"]');\n        if (downloadLink) {…", "filename": ""}], "chunk_position": 7, "heading_path": "Triggering Downloads > Triggering Downloads", "breadcrumbs": "Download Handling in Crawl4AI > Triggering Downloads > Triggering Downloads"}, {"id": "6d4139d2bdc5714c", "url": "https://docs.crawl4ai.com/advanced/file-downloading/", "page_title": "Download Handling in Crawl4AI", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI to handle file downloads during crawling, including enabling downloads, specifying download locations, triggering downloads, and accessing downloaded files.", "heading": "Accessing Downloaded Files", "content": "Page: Download Handling in Crawl4AI\nSection: Accessing Downloaded Files\n\nThe `downloaded_files` attribute of the `CrawlResult` object contains paths to downloaded files.", "code_blocks": [{"language": "", "code": "if result.downloaded_files:\n    print(\"Downloaded files:\")\n    for file_path in result.downloaded_files:\n        print(f\"- {file_path}\")\n        file_size = os.path.getsize(file_path)…", "filename": ""}], "chunk_position": 7, "heading_path": "Accessing Downloaded Files > Accessing Downloaded Files", "breadcrumbs": "Download Handling in Crawl4AI > Accessing Downloaded Files > Accessing Downloaded Files"}, {"id": "6607bd4ac88f260c", "url": "https://docs.crawl4ai.com/advanced/file-downloading/", "page_title": "Download Handling in Crawl4AI", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI to handle file downloads during crawling, including enabling downloads, specifying download locations, triggering downloads, and accessing downloaded files.", "heading": "Example: Downloading Multiple Files", "content": "Page: Download Handling in Crawl4AI\nSection: Example: Downloading Multiple Files\n\n```\nfrom crawl4ai.async_configs import BrowserConfig, CrawlerRunConfig\nimport os\nfrom pathlib import Path\n\nasync def download_multiple_files(url: str, download_path: str):\n    config =…\n```", "code_blocks": [{"language": "", "code": "from crawl4ai.async_configs import BrowserConfig, CrawlerRunConfig\nimport os\nfrom pathlib import Path\n\nasync def download_multiple_files(url: str, download_path: str):\n    config =…", "filename": ""}], "chunk_position": 7, "heading_path": "Example: Downloading Multiple Files > Example: Downloading Multiple Files", "breadcrumbs": "Download Handling in Crawl4AI > Example: Downloading Multiple Files > Example: Downloading Multiple Files"}, {"id": "9ab1a1ed90a72ffa", "url": "https://docs.crawl4ai.com/advanced/file-downloading/", "page_title": "Download Handling in Crawl4AI", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI to handle file downloads during crawling, including enabling downloads, specifying download locations, triggering downloads, and accessing downloaded files.", "heading": "Important Considerations", "content": "Page: Download Handling in Crawl4AI\nSection: Important Considerations\n\n- **Browser Context:**  Downloads are managed within the browser context. Ensure `js_code` correctly targets the download triggers on the webpage.\n\n- **Timing:**  Use `wait_for` in `CrawlerRunConfig`…", "code_blocks": [], "chunk_position": 7, "heading_path": "Important Considerations > Important Considerations", "breadcrumbs": "Download Handling in Crawl4AI > Important Considerations > Important Considerations"}, {"id": "a0d5a5e51a9d86ea", "url": "https://docs.crawl4ai.com/advanced/hooks-auth/", "page_title": "Hooks & Auth in AsyncWebCrawler", "page_type": "guide", "page_summary": "This guide explains how to use hooks in AsyncWebCrawler to customize the crawling pipeline at specific stages, including authentication, route blocking, and pre/post-processing, with a detailed…", "heading": "Introduction", "content": "Page: Hooks & Auth in AsyncWebCrawler\nSection: Introduction\n\n##### Hooks & Auth in AsyncWebCrawler\n\n\n\nCrawl4AI’s  **hooks**  let you customize the crawler at specific points in the pipeline:\n\n\n\n\n1.  **`on_browser_created`**  – After browser creation.\n\n2.…", "code_blocks": [], "chunk_position": 8, "heading_path": "Introduction > Introduction", "breadcrumbs": "Hooks & Auth in AsyncWebCrawler > Introduction > Introduction"}, {"id": "537052351f8cac77", "url": "https://docs.crawl4ai.com/advanced/hooks-auth/", "page_title": "Hooks & Auth in AsyncWebCrawler", "page_type": "guide", "page_summary": "This guide explains how to use hooks in AsyncWebCrawler to customize the crawling pipeline at specific stages, including authentication, route blocking, and pre/post-processing, with a detailed…", "heading": "Example: Using Hooks in AsyncWebCrawler", "content": "Page: Hooks & Auth in AsyncWebCrawler\nSection: Example: Using Hooks in AsyncWebCrawler\n\n```python\nimport asyncio\nimport json\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode\nfrom playwright.async_api import Page, BrowserContext\n\nasync def main():\n    print(\"🔗 Hooks…\n```", "code_blocks": [{"language": "python", "code": "import asyncio\nimport json\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode\nfrom playwright.async_api import Page, BrowserContext\n\nasync def main():\n    print(\"🔗 Hooks…", "filename": ""}], "chunk_position": 8, "heading_path": "Example: Using Hooks in AsyncWebCrawler > Example: Using Hooks in AsyncWebCrawler", "breadcrumbs": "Hooks & Auth in AsyncWebCrawler > Example: Using Hooks in AsyncWebCrawler > Example: Using Hooks in AsyncWebCrawler"}, {"id": "66471252f7d23dc3", "url": "https://docs.crawl4ai.com/advanced/hooks-auth/", "page_title": "Hooks & Auth in AsyncWebCrawler", "page_type": "guide", "page_summary": "This guide explains how to use hooks in AsyncWebCrawler to customize the crawling pipeline at specific stages, including authentication, route blocking, and pre/post-processing, with a detailed…", "heading": "Hook Lifecycle Summary", "content": "Page: Hooks & Auth in AsyncWebCrawler\nSection: Hook Lifecycle Summary\n\n1.  **`on_browser_created`** :\n\n   - Browser is up, but  **no**  pages or contexts yet.\n\n   - Light setup only—don’t try to open or close pages here (that belongs in…", "code_blocks": [], "chunk_position": 8, "heading_path": "Hook Lifecycle Summary > Hook Lifecycle Summary", "breadcrumbs": "Hooks & Auth in AsyncWebCrawler > Hook Lifecycle Summary > Hook Lifecycle Summary"}, {"id": "23cdcaca01d9ccbe", "url": "https://docs.crawl4ai.com/advanced/hooks-auth/", "page_title": "Hooks & Auth in AsyncWebCrawler", "page_type": "guide", "page_summary": "This guide explains how to use hooks in AsyncWebCrawler to customize the crawling pipeline at specific stages, including authentication, route blocking, and pre/post-processing, with a detailed…", "heading": "When to Handle Authentication", "content": "Page: Hooks & Auth in AsyncWebCrawler\nSection: When to Handle Authentication\n\n**Recommended** : Use  **`on_page_context_created`**  if you need to:\n\n\n\n\n- Navigate to a login page or fill forms\n\n- Set cookies or localStorage tokens\n\n- Block resource routes to avoid ads\n\n\n\n\nThis…", "code_blocks": [], "chunk_position": 8, "heading_path": "When to Handle Authentication > When to Handle Authentication", "breadcrumbs": "Hooks & Auth in AsyncWebCrawler > When to Handle Authentication > When to Handle Authentication"}, {"id": "3de5fd09253d5228", "url": "https://docs.crawl4ai.com/advanced/hooks-auth/", "page_title": "Hooks & Auth in AsyncWebCrawler", "page_type": "guide", "page_summary": "This guide explains how to use hooks in AsyncWebCrawler to customize the crawling pipeline at specific stages, including authentication, route blocking, and pre/post-processing, with a detailed…", "heading": "Additional Considerations", "content": "Page: Hooks & Auth in AsyncWebCrawler\nSection: Additional Considerations\n\n- **Session Management** : If you want multiple `arun()` calls to reuse a single session, pass `session_id=` in your `CrawlerRunConfig`. Hooks remain the same.\n\n- **Performance** : Hooks can slow…", "code_blocks": [], "chunk_position": 8, "heading_path": "Additional Considerations > Additional Considerations", "breadcrumbs": "Hooks & Auth in AsyncWebCrawler > Additional Considerations > Additional Considerations"}, {"id": "4301176eb19daea2", "url": "https://docs.crawl4ai.com/advanced/hooks-auth/", "page_title": "Hooks & Auth in AsyncWebCrawler", "page_type": "guide", "page_summary": "This guide explains how to use hooks in AsyncWebCrawler to customize the crawling pipeline at specific stages, including authentication, route blocking, and pre/post-processing, with a detailed…", "heading": "Conclusion", "content": "Page: Hooks & Auth in AsyncWebCrawler\nSection: Conclusion\n\nHooks provide  **fine-grained**  control over:\n\n\n\n\n- **Browser**  creation (light tasks only)\n\n- **Page**  and  **context**  creation (auth, route blocking)\n\n- **Navigation**  phases\n\n- **Final…", "code_blocks": [], "chunk_position": 8, "heading_path": "Conclusion > Conclusion", "breadcrumbs": "Hooks & Auth in AsyncWebCrawler > Conclusion > Conclusion"}, {"id": "bfc31f86990c0ca0", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "Preserve Your Identity with Crawl4AI", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: Preserve Your Identity with Crawl4AI\n\nCrawl4AI empowers you to navigate and interact with the web using your **authentic digital identity**, ensuring you’re recognized as a human and not mistaken for a bot. This tutorial covers:\n\n1.…", "code_blocks": [], "chunk_position": 9, "heading_path": "Preserve Your Identity with Crawl4AI > Preserve Your Identity with Crawl4AI", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > Preserve Your Identity with Crawl4AI > Preserve Your Identity with Crawl4AI"}, {"id": "5746acd1eed2eca9", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "1. Managed Browsers: Your Digital Identity Solution", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 1. Managed Browsers: Your Digital Identity Solution\n\n**Managed Browsers** let developers create and use **persistent browser profiles**. These profiles store local storage, cookies, and other session data, letting you browse as your **real self**…", "code_blocks": [], "chunk_position": 9, "heading_path": "1. Managed Browsers: Your Digital Identity Solution > 1. Managed Browsers: Your Digital Identity Solution", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > 1. Managed Browsers: Your Digital Identity Solution > 1. Managed Browsers: Your Digital Identity Solution"}, {"id": "484a0d1add597094", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "Key Benefits", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: Key Benefits\n\n- **Authentic Browsing Experience**: Retain session data and browser fingerprints as though you’re a normal user.\n- **Effortless Configuration**: Once you log in or solve CAPTCHAs in your chosen data…", "code_blocks": [], "chunk_position": 9, "heading_path": "Key Benefits > Key Benefits", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > Key Benefits > Key Benefits"}, {"id": "1d85a04347b88755", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "Creating a User Data Directory (Command-Line Approach via Playwright)", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: Creating a User Data Directory (Command-Line Approach via Playwright)\n\nIf you installed Crawl4AI (which installs Playwright under the hood), you already have a Playwright-managed Chromium on your system. Follow these steps to launch that **Chromium** from your command…", "code_blocks": [{"language": "bash", "code": "python -m playwright install --dry-run", "filename": ""}, {"language": "bash", "code": "playwright install --dry-run", "filename": ""}, {"language": "bash", "code": "~/.cache/ms-playwright/chromium-1234/chrome-linux/chrome", "filename": ""}, {"language": "bash", "code": "# Linux example\n~/.cache/ms-playwright/chromium-1234/chrome-linux/chrome \\\n    --user-data-dir=/home/<you>/my_chrome_profile", "filename": ""}, {"language": "bash", "code": "# macOS example (Playwright’s internal binary)\n~/Library/Caches/ms-playwright/chromium-1234/chrome-mac/Chromium.app/Contents/MacOS/Chromium \\\n    --user-data-dir=/Users/<you>/my_chrome_profile", "filename": ""}, {"language": "powershell", "code": "# Windows example (PowerShell/cmd)\n\"C:\\Users\\<you>\\AppData\\Local\\ms-playwright\\chromium-1234\\chrome-win\\chrome.exe\" ^\n    --user-data-dir=\"C:\\Users\\<you>\\my_chrome_profile\"", "filename": ""}, {"language": "python", "code": "from crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig\n\nbrowser_config = BrowserConfig(\n    headless=True,\n    use_managed_browser=True,…", "filename": ""}], "chunk_position": 9, "heading_path": "Creating a User Data Directory (Command-Line Approach via Playwright) > Creating a User Data Directory (Command-Line Approach via Playwright)", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > Creating a User Data Directory (Command-Line Approach via Playwright) > Creating a User Data Directory (Command-Line Approach via Playwright)"}, {"id": "15850854737ec2d9", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "Creating a Profile Using the Crawl4AI CLI (Easiest)", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: Creating a Profile Using the Crawl4AI CLI (Easiest)\n\nIf you prefer a guided, interactive setup, use the built-in CLI to create and manage persistent browser profiles.\n\n1. Launch the profile manager (`crwl profiles`).\n\n2. Choose \"Create new profile\" and…", "code_blocks": [{"language": "bash", "code": "crwl profiles", "filename": ""}, {"language": "python", "code": "from crawl4ai import AsyncWebCrawler, BrowserConfig\n\nprofile_path = \"/home/<you>/.crawl4ai/profiles/test_profile_1\"\n\nbrowser_config = BrowserConfig(\n    headless=True,\n    use_managed_browser=True,…", "filename": ""}], "chunk_position": 9, "heading_path": "Creating a Profile Using the Crawl4AI CLI (Easiest) > Creating a Profile Using the Crawl4AI CLI (Easiest)", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > Creating a Profile Using the Crawl4AI CLI (Easiest) > Creating a Profile Using the Crawl4AI CLI (Easiest)"}, {"id": "125706a0436ac398", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "3. Using Managed Browsers in Crawl4AI", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 3. Using Managed Browsers in Crawl4AI\n\nOnce you have a data directory with your session data, pass it to **`BrowserConfig`**:", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig\n\nasync def main():\n    # 1) Reference your persistent data directory\n    browser_config = BrowserConfig(…", "filename": ""}], "chunk_position": 9, "heading_path": "3. Using Managed Browsers in Crawl4AI > 3. Using Managed Browsers in Crawl4AI", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > 3. Using Managed Browsers in Crawl4AI > 3. Using Managed Browsers in Crawl4AI"}, {"id": "7c1a3b7b6ea78a45", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "Workflow", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: Workflow\n\n1. **Login** externally (via CLI or your normal Chrome with `--user-data-dir=...`).\n2. **Close** that browser.\n3. **Use** the same folder in `user_data_dir=` in Crawl4AI.\n4. **Crawl** – The site sees…", "code_blocks": [], "chunk_position": 9, "heading_path": "Workflow > Workflow", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > Workflow > Workflow"}, {"id": "0392c4da05929706", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "4. Magic Mode: Simplified Automation", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 4. Magic Mode: Simplified Automation\n\nIf you **don’t** need a persistent profile or identity-based approach, **Magic Mode** offers a quick way to simulate human-like browsing without storing long-term data.\n\n**Magic Mode**:\n\n- Simulates…", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nasync with AsyncWebCrawler() as crawler:\n    result = await crawler.arun(\n        url=\"https://example.com\",\n        config=CrawlerRunConfig(…", "filename": ""}], "chunk_position": 9, "heading_path": "4. Magic Mode: Simplified Automation > 4. Magic Mode: Simplified Automation", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > 4. Magic Mode: Simplified Automation > 4. Magic Mode: Simplified Automation"}, {"id": "390f67992e62a961", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "5. Comparing Managed Browsers vs. Magic Mode", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 5. Comparing Managed Browsers vs. Magic Mode\n\n| Feature | **Managed Browsers** | **Magic Mode** |\n| --- | --- | --- |\n| **Session Persistence** | Full localStorage/cookies retained in user_data_dir | No persistent data (fresh each run) |\n|…", "code_blocks": [], "chunk_position": 9, "heading_path": "5. Comparing Managed Browsers vs. Magic Mode > 5. Comparing Managed Browsers vs. Magic Mode", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > 5. Comparing Managed Browsers vs. Magic Mode > 5. Comparing Managed Browsers vs. Magic Mode"}, {"id": "51d63497e6349f5c", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "6. Using the BrowserProfiler Class", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 6. Using the BrowserProfiler Class\n\nCrawl4AI provides a dedicated `BrowserProfiler` class for managing browser profiles, making it easy to create, list, and delete profiles for identity-based browsing.", "code_blocks": [], "chunk_position": 9, "heading_path": "6. Using the BrowserProfiler Class > 6. Using the BrowserProfiler Class", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > 6. Using the BrowserProfiler Class > 6. Using the BrowserProfiler Class"}, {"id": "ba4ef02f01b04f82", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "Creating and Managing Profiles with BrowserProfiler", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: Creating and Managing Profiles with BrowserProfiler\n\nThe `BrowserProfiler` class offers a comprehensive API for browser profile management:\n\n**How profile creation works:**\n1. A browser window opens for you to interact with\n2. You log in to websites,…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import BrowserProfiler\n\nasync def manage_profiles():\n    # Create a profiler instance\n    profiler = BrowserProfiler()\n\n    # Create a profile interactively - opens a…", "filename": ""}], "chunk_position": 9, "heading_path": "Creating and Managing Profiles with BrowserProfiler > Creating and Managing Profiles with BrowserProfiler", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > Creating and Managing Profiles with BrowserProfiler > Creating and Managing Profiles with BrowserProfiler"}, {"id": "61cb6cfdba7ed7cc", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "Interactive Profile Management", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: Interactive Profile Management\n\nThe `BrowserProfiler` also offers an interactive management console that guides you through profile creation, listing, and deletion:", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import BrowserProfiler, AsyncWebCrawler, BrowserConfig\n\n# Define a function to use a profile for crawling\nasync def crawl_with_profile(profile_path, url):…", "filename": ""}], "chunk_position": 9, "heading_path": "Interactive Profile Management > Interactive Profile Management", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > Interactive Profile Management > Interactive Profile Management"}, {"id": "3ac103efa48b7364", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "Legacy Methods", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: Legacy Methods\n\nFor backward compatibility, the previous methods on `ManagedBrowser` are still available, but they delegate to the new `BrowserProfiler` class:", "code_blocks": [{"language": "python", "code": "from crawl4ai.browser_manager import ManagedBrowser\n\n# These methods still work but use BrowserProfiler internally\nprofiles = ManagedBrowser.list_profiles()", "filename": ""}], "chunk_position": 9, "heading_path": "Legacy Methods > Legacy Methods", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > Legacy Methods > Legacy Methods"}, {"id": "048596e7bbc6edf8", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "Complete Example", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: Complete Example\n\nSee the full example in `docs/examples/identity_based_browsing.py` for a complete demonstration of creating and using profiles for authenticated browsing using the new `BrowserProfiler` class.", "code_blocks": [], "chunk_position": 9, "heading_path": "Complete Example > Complete Example", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > Complete Example > Complete Example"}, {"id": "8a26a8297453fd6b", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "7. Locale, Timezone, and Geolocation Control", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 7. Locale, Timezone, and Geolocation Control\n\nIn addition to using persistent profiles, Crawl4AI supports customizing your browser's locale, timezone, and geolocation settings. These features enhance your identity-based browsing experience by…", "code_blocks": [], "chunk_position": 9, "heading_path": "7. Locale, Timezone, and Geolocation Control > 7. Locale, Timezone, and Geolocation Control", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > 7. Locale, Timezone, and Geolocation Control > 7. Locale, Timezone, and Geolocation Control"}, {"id": "d5e1071c64bfa8ea", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "Setting Locale and Timezone", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: Setting Locale and Timezone\n\nYou can set the browser's locale and timezone through `CrawlerRunConfig`:\n\n**How it works:**\n- `locale` affects language preferences, date formats, number formats, etc.\n- `timezone_id` affects…", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nasync with AsyncWebCrawler() as crawler:\n    result = await crawler.arun(\n        url=\"https://example.com\",\n        config=CrawlerRunConfig(…", "filename": ""}], "chunk_position": 9, "heading_path": "Setting Locale and Timezone > Setting Locale and Timezone", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > Setting Locale and Timezone > Setting Locale and Timezone"}, {"id": "5300eb2de5e307d4", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "Configuring Geolocation", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: Configuring Geolocation\n\nControl the GPS coordinates reported by the browser's geolocation API:\n\n**Important notes:**\n- When `geolocation` is specified, the browser is automatically granted permission to access location\n-…", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler, CrawlerRunConfig, GeolocationConfig\n\nasync with AsyncWebCrawler() as crawler:\n    result = await crawler.arun(\n        url=\"https://maps.google.com\",  # Or any…", "filename": ""}], "chunk_position": 9, "heading_path": "Configuring Geolocation > Configuring Geolocation", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > Configuring Geolocation > Configuring Geolocation"}, {"id": "492484f38ecdaca3", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "Combining with Managed Browsers", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: Combining with Managed Browsers\n\nThese settings work perfectly with managed browsers for a complete identity solution:\n\nCombining persistent profiles with precise geolocation and region settings gives you complete control over your…", "code_blocks": [{"language": "python", "code": "from crawl4ai import (\n    AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, \n    GeolocationConfig\n)\n\nbrowser_config = BrowserConfig(\n    use_managed_browser=True,…", "filename": ""}], "chunk_position": 9, "heading_path": "Combining with Managed Browsers > Combining with Managed Browsers", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > Combining with Managed Browsers > Combining with Managed Browsers"}, {"id": "c88839e8049f9a3a", "url": "https://docs.crawl4ai.com/advanced/identity-based-crawling/", "page_title": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to use Crawl4AI's Managed Browsers and BrowserProfiler to preserve a user's authentic digital identity while crawling, including persistent profiles, Magic Mode fallback, and…", "heading": "8. Summary", "content": "Page: Identity Based Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 8. Summary\n\n- **Create** your user-data directory either:\n  - By launching Chrome/Chromium externally with `--user-data-dir=/some/path`\n  - Or by using the built-in `BrowserProfiler.create_profile()` method\n  -…", "code_blocks": [], "chunk_position": 9, "heading_path": "8. Summary > 8. Summary", "breadcrumbs": "Identity Based Crawling - Crawl4AI Documentation (v0.9.x) > 8. Summary > 8. Summary"}, {"id": "031196321f1e4506", "url": "https://docs.crawl4ai.com/advanced/lazy-loading/", "page_title": "Lazy Loading - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to handle lazy-loaded images in Crawl4AI by using wait_for_images, scan_full_page, and scroll_delay settings, and how to combine these with media filters and domain exclusions.", "heading": "Handling Lazy-Loaded Images", "content": "Page: Lazy Loading - Crawl4AI Documentation (v0.9.x)\nSection: Handling Lazy-Loaded Images\n\nMany websites now load images **lazily** as you scroll. If you need to ensure they appear in your final crawl (and in `result.media`), consider:\n\n1. **`wait_for_images=True`** – Wait for images to…", "code_blocks": [], "chunk_position": 10, "heading_path": "Handling Lazy-Loaded Images > Handling Lazy-Loaded Images", "breadcrumbs": "Lazy Loading - Crawl4AI Documentation (v0.9.x) > Handling Lazy-Loaded Images > Handling Lazy-Loaded Images"}, {"id": "e36bcf64da33bb76", "url": "https://docs.crawl4ai.com/advanced/lazy-loading/", "page_title": "Lazy Loading - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to handle lazy-loaded images in Crawl4AI by using wait_for_images, scan_full_page, and scroll_delay settings, and how to combine these with media filters and domain exclusions.", "heading": "Example: Ensuring Lazy Images Appear", "content": "Page: Lazy Loading - Crawl4AI Documentation (v0.9.x)\nSection: Example: Ensuring Lazy Images Appear\n\n**Explanation**:\n\n- **`wait_for_images=True`**\n\n  The crawler tries to ensure images have finished loading before finalizing the HTML.\n\n- **`scan_full_page=True`**\n\n  Tells the crawler to attempt…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, BrowserConfig\nfrom crawl4ai.async_configs import CacheMode\n\nasync def main():\n    config = CrawlerRunConfig(\n        # Force the…", "filename": ""}], "chunk_position": 10, "heading_path": "Example: Ensuring Lazy Images Appear > Example: Ensuring Lazy Images Appear", "breadcrumbs": "Lazy Loading - Crawl4AI Documentation (v0.9.x) > Example: Ensuring Lazy Images Appear > Example: Ensuring Lazy Images Appear"}, {"id": "4f7a81fec779bcfb", "url": "https://docs.crawl4ai.com/advanced/lazy-loading/", "page_title": "Lazy Loading - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to handle lazy-loaded images in Crawl4AI by using wait_for_images, scan_full_page, and scroll_delay settings, and how to combine these with media filters and domain exclusions.", "heading": "Combining with Other Link & Media Filters", "content": "Page: Lazy Loading - Crawl4AI Documentation (v0.9.x)\nSection: Combining with Other Link & Media Filters\n\nYou can still combine **lazy-load** logic with the usual **exclude_external_images**, **exclude_domains**, or link filtration:\n\nThis approach ensures you see **all** images from the main domain while…", "code_blocks": [{"language": "python", "code": "config = CrawlerRunConfig(\n    wait_for_images=True,\n    scan_full_page=True,\n    scroll_delay=0.5,\n\n    # Filter out external images if you only want local ones\n    exclude_external_images=True,…", "filename": ""}], "chunk_position": 10, "heading_path": "Combining with Other Link & Media Filters > Combining with Other Link & Media Filters", "breadcrumbs": "Lazy Loading - Crawl4AI Documentation (v0.9.x) > Combining with Other Link & Media Filters > Combining with Other Link & Media Filters"}, {"id": "f3f06a7992462338", "url": "https://docs.crawl4ai.com/advanced/lazy-loading/", "page_title": "Lazy Loading - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This guide explains how to handle lazy-loaded images in Crawl4AI by using wait_for_images, scan_full_page, and scroll_delay settings, and how to combine these with media filters and domain exclusions.", "heading": "Tips & Troubleshooting", "content": "Page: Lazy Loading - Crawl4AI Documentation (v0.9.x)\nSection: Tips & Troubleshooting\n\n1. **Long Pages**\n\n   - Setting `scan_full_page=True` on extremely long or infinite-scroll pages can be resource-intensive.\n\n   - Consider using [hooks](../../core/page-interaction/) or specialized…", "code_blocks": [], "chunk_position": 10, "heading_path": "Tips & Troubleshooting > Tips & Troubleshooting", "breadcrumbs": "Lazy Loading - Crawl4AI Documentation (v0.9.x) > Tips & Troubleshooting > Tips & Troubleshooting"}, {"id": "ca6f956773888071", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "1. Introduction", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: 1. Introduction\n\n> **Heads Up** : Crawl4AI supports advanced dispatchers for **parallel** or **throttled** crawling, providing dynamic rate limiting and memory usage checks. The built-in `arun_many()` function uses…", "code_blocks": [], "chunk_position": 11, "heading_path": "1. Introduction > 1. Introduction", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > 1. Introduction > 1. Introduction"}, {"id": "e02a533eb22dde95", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "2.1 Rate Limiter", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: 2.1 Rate Limiter\n\nHere’s the revised and simplified explanation of the **RateLimiter** , focusing on constructor parameters and adhering to your markdown style and mkDocs guidelines.", "code_blocks": [{"language": "python", "code": "class RateLimiter:\n    def __init__(\n        # Random delay range between requests\n        base_delay: Tuple[float, float] = (1.0, 3.0),  \n\n        # Maximum backoff delay\n        max_delay: float =…", "filename": ""}], "chunk_position": 11, "heading_path": "2.1 Rate Limiter > 2.1 Rate Limiter", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > 2.1 Rate Limiter > 2.1 Rate Limiter"}, {"id": "f5d5731e37356fe9", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "RateLimiter Constructor Parameters", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: RateLimiter Constructor Parameters\n\nThe **RateLimiter** is a utility that helps manage the pace of requests to avoid overloading servers or getting blocked due to rate limits. It operates internally to delay requests and handle retries…", "code_blocks": [{"language": "python", "code": "from crawl4ai import RateLimiter\n\n# Create a RateLimiter with custom settings\nrate_limiter = RateLimiter(\n    base_delay=(2.0, 4.0),  # Random delay between 2-4 seconds\n    max_delay=30.0,         #…", "filename": ""}], "chunk_position": 11, "heading_path": "RateLimiter Constructor Parameters > RateLimiter Constructor Parameters", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > RateLimiter Constructor Parameters > RateLimiter Constructor Parameters"}, {"id": "c15645890a8c05cb", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "2.2 Crawler Monitor", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: 2.2 Crawler Monitor\n\nThe CrawlerMonitor provides real-time visibility into crawling operations:\n\n **Display Modes** :\n\n- **DETAILED** : Shows individual task status, memory usage, and timing\n- **AGGREGATED** : Displays…", "code_blocks": [{"language": "python", "code": "from crawl4ai import CrawlerMonitor, DisplayMode\nmonitor = CrawlerMonitor(\n    # Maximum rows in live display\n    max_visible_rows=15,          \n\n    # DETAILED or AGGREGATED view…", "filename": ""}], "chunk_position": 11, "heading_path": "2.2 Crawler Monitor > 2.2 Crawler Monitor", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > 2.2 Crawler Monitor > 2.2 Crawler Monitor"}, {"id": "0fbcd30842d71ba4", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "3.1 MemoryAdaptiveDispatcher (Default)", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: 3.1 MemoryAdaptiveDispatcher (Default)\n\nAutomatically manages concurrency based on system memory usage:\n\n **Constructor Parameters:** \n\n1. **`memory_threshold_percent`** (`float`, default: `90.0`)\n\n   Specifies the memory usage threshold…", "code_blocks": [{"language": "python", "code": "from crawl4ai.async_dispatcher import MemoryAdaptiveDispatcher\n\ndispatcher = MemoryAdaptiveDispatcher(\n    memory_threshold_percent=90.0,  # Pause if memory exceeds this\n    check_interval=1.0,…", "filename": ""}], "chunk_position": 11, "heading_path": "3.1 MemoryAdaptiveDispatcher (Default) > 3.1 MemoryAdaptiveDispatcher (Default)", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > 3.1 MemoryAdaptiveDispatcher (Default) > 3.1 MemoryAdaptiveDispatcher (Default)"}, {"id": "4ba8c8175b731d9d", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "3.2 SemaphoreDispatcher", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: 3.2 SemaphoreDispatcher\n\nProvides simple concurrency control with a fixed limit:\n\n **Constructor Parameters:** \n\n1. **`max_session_permit`** (`int`, default: `20`)\n\n   The maximum number of concurrent crawling tasks allowed,…", "code_blocks": [{"language": "python", "code": "from crawl4ai.async_dispatcher import SemaphoreDispatcher\n\ndispatcher = SemaphoreDispatcher(\n    max_session_permit=20,         # Maximum concurrent tasks\n    rate_limiter=RateLimiter(      #…", "filename": ""}], "chunk_position": 11, "heading_path": "3.2 SemaphoreDispatcher > 3.2 SemaphoreDispatcher", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > 3.2 SemaphoreDispatcher > 3.2 SemaphoreDispatcher"}, {"id": "c7ec6b7b805a5514", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "4.1 Batch Processing (Default)", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: 4.1 Batch Processing (Default)\n\n**Review:** \n\n- **Purpose:** Executes a batch crawl with all URLs processed together after crawling is complete.\n- **Dispatcher:** Uses `MemoryAdaptiveDispatcher` to manage concurrency and system…", "code_blocks": [{"language": "python", "code": "async def crawl_batch():\n    browser_config = BrowserConfig(headless=True, verbose=False)\n    run_config = CrawlerRunConfig(\n        cache_mode=CacheMode.BYPASS,\n        stream=False  # Default: get…", "filename": ""}], "chunk_position": 11, "heading_path": "4.1 Batch Processing (Default) > 4.1 Batch Processing (Default)", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > 4.1 Batch Processing (Default) > 4.1 Batch Processing (Default)"}, {"id": "8b46a9782df96c1b", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "4.2 Streaming Mode", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: 4.2 Streaming Mode\n\n**Review:** \n\n- **Purpose:** Enables streaming to process results as soon as they’re available.\n- **Dispatcher:** Uses `MemoryAdaptiveDispatcher` for concurrency and memory management.\n- **Stream:**…", "code_blocks": [{"language": "python", "code": "async def crawl_streaming():\n    browser_config = BrowserConfig(headless=True, verbose=False)\n    run_config = CrawlerRunConfig(\n        cache_mode=CacheMode.BYPASS,\n        stream=True  # Enable…", "filename": ""}], "chunk_position": 11, "heading_path": "4.2 Streaming Mode > 4.2 Streaming Mode", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > 4.2 Streaming Mode > 4.2 Streaming Mode"}, {"id": "f0eaca03d74895d1", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "4.3 Semaphore-based Crawling", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: 4.3 Semaphore-based Crawling\n\n**Review:** \n\n- **Purpose:** Uses `SemaphoreDispatcher` to limit concurrency with a fixed number of slots.\n- **Dispatcher:** Configured with a semaphore to control parallel crawling tasks.\n- **Rate…", "code_blocks": [{"language": "python", "code": "async def crawl_with_semaphore(urls):\n    browser_config = BrowserConfig(headless=True, verbose=False)\n    run_config = CrawlerRunConfig(cache_mode=CacheMode.BYPASS)\n\n    dispatcher =…", "filename": ""}], "chunk_position": 11, "heading_path": "4.3 Semaphore-based Crawling > 4.3 Semaphore-based Crawling", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > 4.3 Semaphore-based Crawling > 4.3 Semaphore-based Crawling"}, {"id": "bdf8644ff2f3db38", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "4.4 Robots.txt Consideration", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: 4.4 Robots.txt Consideration\n\n**Review:** \n\n- **Purpose:** Ensures compliance with `robots.txt` rules for ethical and legal web crawling.\n- **Configuration:** Set `check_robots_txt=True` to validate each URL against `robots.txt`…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\n\nasync def main():\n    urls = [\n        \"https://example1.com\",\n        \"https://example2.com\",…", "filename": ""}], "chunk_position": 11, "heading_path": "4.4 Robots.txt Consideration > 4.4 Robots.txt Consideration", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > 4.4 Robots.txt Consideration > 4.4 Robots.txt Consideration"}, {"id": "3add42b774c69077", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "5. Dispatch Results", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: 5. Dispatch Results\n\nEach crawl result includes dispatch information:\n\nAccess via `result.dispatch_result`:", "code_blocks": [{"language": "python", "code": "@dataclass\nclass DispatchResult:\n    task_id: str\n    memory_usage: float\n    peak_memory: float\n    start_time: datetime\n    end_time: datetime\n    error_message: str = \"\"\nCopy", "filename": ""}, {"language": "python", "code": "for result in results:\n    if result.success:\n        dr = result.dispatch_result\n        print(f\"URL: {result.url}\")\n        print(f\"Memory: {dr.memory_usage:.1f}MB\")\n        print(f\"Duration:…", "filename": ""}], "chunk_position": 11, "heading_path": "5. Dispatch Results > 5. Dispatch Results", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > 5. Dispatch Results > 5. Dispatch Results"}, {"id": "018349e3f9e40b6a", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "6. URL-Specific Configurations", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: 6. URL-Specific Configurations\n\nWhen crawling diverse content types, you often need different configurations for different URLs. For example:\n- PDFs need specialized extraction\n- Blog pages benefit from content filtering\n- Dynamic…", "code_blocks": [], "chunk_position": 11, "heading_path": "6. URL-Specific Configurations > 6. URL-Specific Configurations", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > 6. URL-Specific Configurations > 6. URL-Specific Configurations"}, {"id": "77cb4dba88676779", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "6.1 Basic URL Pattern Matching", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: 6.1 Basic URL Pattern Matching\n\n**Important** : A `CrawlerRunConfig` without `url_matcher` (or with `url_matcher=None`) matches ALL URLs. This makes it perfect as a default/fallback configuration.\n\nThe `url_matcher` parameter…", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler, CrawlerRunConfig, MatchMode\nfrom crawl4ai.processors.pdf import PDFContentScrapingStrategy\nfrom crawl4ai.extraction_strategy import…", "filename": ""}], "chunk_position": 11, "heading_path": "6.1 Basic URL Pattern Matching > 6.1 Basic URL Pattern Matching", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > 6.1 Basic URL Pattern Matching > 6.1 Basic URL Pattern Matching"}, {"id": "58c574bd9479877f", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "6.2 Advanced Pattern Matching", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: 6.2 Advanced Pattern Matching\n\n**Important** : A `CrawlerRunConfig` without `url_matcher` (or with `url_matcher=None`) matches ALL URLs. This makes it perfect as a default/fallback configuration.\n\nThe `url_matcher` parameter…", "code_blocks": [], "chunk_position": 11, "heading_path": "6.2 Advanced Pattern Matching > 6.2 Advanced Pattern Matching", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > 6.2 Advanced Pattern Matching > 6.2 Advanced Pattern Matching"}, {"id": "44652579ae908911", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "Glob Patterns (Strings)", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: Glob Patterns (Strings)\n\n```python\n# Simple patterns\n\"*.pdf\"                    # Any PDF file\n\"*/api/*\"                  # Any URL with /api/ in path\n\"https://*.example.com/*\"  # Subdomain matching\n\"*://example.com/blog/*\"   # Any…\n```", "code_blocks": [{"language": "python", "code": "# Simple patterns\n\"*.pdf\"                    # Any PDF file\n\"*/api/*\"                  # Any URL with /api/ in path\n\"https://*.example.com/*\"  # Subdomain matching\n\"*://example.com/blog/*\"   # Any…", "filename": ""}], "chunk_position": 11, "heading_path": "Glob Patterns (Strings) > Glob Patterns (Strings)", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > Glob Patterns (Strings) > Glob Patterns (Strings)"}, {"id": "fdc731646a9ffc39", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "Custom Functions", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: Custom Functions\n\n```python\n# Complex logic with lambdas\nlambda url: url.startswith('https://') and 'secure' in url\nlambda url: len(url) > 50 and url.count('/') > 5\nlambda url: any(domain in url for domain in ['api.', 'data.',…\n```", "code_blocks": [{"language": "python", "code": "# Complex logic with lambdas\nlambda url: url.startswith('https://') and 'secure' in url\nlambda url: len(url) > 50 and url.count('/') > 5\nlambda url: any(domain in url for domain in ['api.', 'data.',…", "filename": ""}], "chunk_position": 11, "heading_path": "Custom Functions > Custom Functions", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > Custom Functions > Custom Functions"}, {"id": "83c55e38e1bbe71b", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "Mixed Lists with AND/OR Logic", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: Mixed Lists with AND/OR Logic\n\n```python\n# Combine multiple conditions\nCrawlerRunConfig(\n    url_matcher=[\n        \"https://*\",                        # Must be HTTPS\n        lambda url: 'internal' in url,      # Must contain 'internal'…\n```", "code_blocks": [{"language": "python", "code": "# Combine multiple conditions\nCrawlerRunConfig(\n    url_matcher=[\n        \"https://*\",                        # Must be HTTPS\n        lambda url: 'internal' in url,      # Must contain 'internal'…", "filename": ""}], "chunk_position": 11, "heading_path": "Mixed Lists with AND/OR Logic > Mixed Lists with AND/OR Logic", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > Mixed Lists with AND/OR Logic > Mixed Lists with AND/OR Logic"}, {"id": "e9bdbe54eed70527", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "6.3 Practical Example: News Site Crawler", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: 6.3 Practical Example: News Site Crawler\n\n```python\nasync def crawl_news_site():\n    dispatcher = MemoryAdaptiveDispatcher(\n        memory_threshold_percent=70.0,\n        rate_limiter=RateLimiter(base_delay=(1.0, 2.0))\n    )\n\n    configs = [\n        #…\n```", "code_blocks": [{"language": "python", "code": "async def crawl_news_site():\n    dispatcher = MemoryAdaptiveDispatcher(\n        memory_threshold_percent=70.0,\n        rate_limiter=RateLimiter(base_delay=(1.0, 2.0))\n    )\n\n    configs = [\n        #…", "filename": ""}], "chunk_position": 11, "heading_path": "6.3 Practical Example: News Site Crawler > 6.3 Practical Example: News Site Crawler", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > 6.3 Practical Example: News Site Crawler > 6.3 Practical Example: News Site Crawler"}, {"id": "35acba2b676faa0a", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "6.4 Best Practices", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: 6.4 Best Practices\n\n- **Order Matters** : Configs are evaluated in order - put specific patterns before general ones\n- **Default Config Behavior** :\n  - A config without `url_matcher` matches ALL URLs\n  - Always include…", "code_blocks": [{"language": "python", "code": "config = CrawlerRunConfig(url_matcher=\"*.pdf\")\nprint(config.is_match(\"https://example.com/doc.pdf\"))  # True\n\ndefault_config = CrawlerRunConfig()  # No…", "filename": ""}], "chunk_position": 11, "heading_path": "6.4 Best Practices > 6.4 Best Practices", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > 6.4 Best Practices > 6.4 Best Practices"}, {"id": "8188cc8e6c1e21c1", "url": "https://docs.crawl4ai.com/advanced/multi-url-crawling/", "page_title": "Advanced Multi-URL Crawling with Dispatchers", "page_type": "api", "page_summary": "This page covers advanced multi-URL crawling using dispatchers in Crawl4AI, including RateLimiter, CrawlerMonitor, MemoryAdaptiveDispatcher, and SemaphoreDispatcher, with usage examples and…", "heading": "7. Summary", "content": "Page: Advanced Multi-URL Crawling with Dispatchers\nSection: 7. Summary\n\n1. **Two Dispatcher Types** :\n\n   - MemoryAdaptiveDispatcher (default): Dynamic concurrency based on memory\n   - SemaphoreDispatcher: Fixed concurrency limit\n\n2. **Optional Components** :\n\n   -…", "code_blocks": [], "chunk_position": 11, "heading_path": "7. Summary > 7. Summary", "breadcrumbs": "Advanced Multi-URL Crawling with Dispatchers > 7. Summary > 7. Summary"}, {"id": "15952a2ce46ffb14", "url": "https://docs.crawl4ai.com/advanced/network-console-capture/", "page_title": "Network Requests & Console Message Capturing", "page_type": "guide", "page_summary": "This page explains how to capture network requests and browser console messages during a crawl using Crawl4AI, including configuration, example usage, data structures, benefits, and use cases.", "heading": "Configuration", "content": "Page: Network Requests & Console Message Capturing\nSection: Configuration\n\nTo enable network and console capturing, use these configuration options:", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\n# Enable both network request capture and console message capture\nconfig = CrawlerRunConfig(\n    capture_network_requests=True,  # Capture all…", "filename": ""}], "chunk_position": 12, "heading_path": "Configuration > Configuration", "breadcrumbs": "Network Requests & Console Message Capturing > Configuration > Configuration"}, {"id": "f482fbff4253e9c3", "url": "https://docs.crawl4ai.com/advanced/network-console-capture/", "page_title": "Network Requests & Console Message Capturing", "page_type": "guide", "page_summary": "This page explains how to capture network requests and browser console messages during a crawl using Crawl4AI, including configuration, example usage, data structures, benefits, and use cases.", "heading": "Example Usage", "content": "Page: Network Requests & Console Message Capturing\nSection: Example Usage\n\n```python\nimport asyncio\nimport json\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nasync def main():\n    # Enable both network request capture and console message capture\n    config =…\n```", "code_blocks": [{"language": "python", "code": "import asyncio\nimport json\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nasync def main():\n    # Enable both network request capture and console message capture\n    config =…", "filename": ""}], "chunk_position": 12, "heading_path": "Example Usage > Example Usage", "breadcrumbs": "Network Requests & Console Message Capturing > Example Usage > Example Usage"}, {"id": "873b5abc1f95bbbf", "url": "https://docs.crawl4ai.com/advanced/network-console-capture/", "page_title": "Network Requests & Console Message Capturing", "page_type": "guide", "page_summary": "This page explains how to capture network requests and browser console messages during a crawl using Crawl4AI, including configuration, example usage, data structures, benefits, and use cases.", "heading": "Captured Data Structure", "content": "Page: Network Requests & Console Message Capturing\nSection: Captured Data Structure\n\nThe `result.network_requests` contains a list of dictionaries, each representing a network event with these common fields:\n\n| Field | Description |\n| --- | --- |\n| `event_type` | Type of event:…", "code_blocks": [], "chunk_position": 12, "heading_path": "Captured Data Structure > Captured Data Structure", "breadcrumbs": "Network Requests & Console Message Capturing > Captured Data Structure > Captured Data Structure"}, {"id": "357405fb9eafca61", "url": "https://docs.crawl4ai.com/advanced/network-console-capture/", "page_title": "Network Requests & Console Message Capturing", "page_type": "guide", "page_summary": "This page explains how to capture network requests and browser console messages during a crawl using Crawl4AI, including configuration, example usage, data structures, benefits, and use cases.", "heading": "Network Requests", "content": "Page: Network Requests & Console Message Capturing\nSection: Network Requests\n\nThe `result.network_requests` contains a list of dictionaries, each representing a network event with these common fields:\n\n| Field | Description |\n| --- | --- |\n| `event_type` | Type of event:…", "code_blocks": [], "chunk_position": 12, "heading_path": "Network Requests > Network Requests", "breadcrumbs": "Network Requests & Console Message Capturing > Network Requests > Network Requests"}, {"id": "59328f687ab30c89", "url": "https://docs.crawl4ai.com/advanced/network-console-capture/", "page_title": "Network Requests & Console Message Capturing", "page_type": "guide", "page_summary": "This page explains how to capture network requests and browser console messages during a crawl using Crawl4AI, including configuration, example usage, data structures, benefits, and use cases.", "heading": "Request Event Fields", "content": "Page: Network Requests & Console Message Capturing\nSection: Request Event Fields\n\n```json\n{\n  \"event_type\": \"request\",\n  \"url\": \"https://example.com/api/data.json\",\n  \"method\": \"GET\",\n  \"headers\": {\"User-Agent\": \"...\", \"Accept\": \"...\"},\n  \"post_data\": \"key=value&otherkey=value\",…\n```", "code_blocks": [{"language": "json", "code": "{\n  \"event_type\": \"request\",\n  \"url\": \"https://example.com/api/data.json\",\n  \"method\": \"GET\",\n  \"headers\": {\"User-Agent\": \"...\", \"Accept\": \"...\"},\n  \"post_data\": \"key=value&otherkey=value\",…", "filename": ""}], "chunk_position": 12, "heading_path": "Request Event Fields > Request Event Fields", "breadcrumbs": "Network Requests & Console Message Capturing > Request Event Fields > Request Event Fields"}, {"id": "d919d4953ce3b224", "url": "https://docs.crawl4ai.com/advanced/network-console-capture/", "page_title": "Network Requests & Console Message Capturing", "page_type": "guide", "page_summary": "This page explains how to capture network requests and browser console messages during a crawl using Crawl4AI, including configuration, example usage, data structures, benefits, and use cases.", "heading": "Response Event Fields", "content": "Page: Network Requests & Console Message Capturing\nSection: Response Event Fields\n\n```json\n{\n  \"event_type\": \"response\",\n  \"url\": \"https://example.com/api/data.json\",\n  \"status\": 200,\n  \"status_text\": \"OK\",\n  \"headers\": {\"Content-Type\": \"application/json\", \"Cache-Control\": \"...\"},…\n```", "code_blocks": [{"language": "json", "code": "{\n  \"event_type\": \"response\",\n  \"url\": \"https://example.com/api/data.json\",\n  \"status\": 200,\n  \"status_text\": \"OK\",\n  \"headers\": {\"Content-Type\": \"application/json\", \"Cache-Control\": \"...\"},…", "filename": ""}], "chunk_position": 12, "heading_path": "Response Event Fields > Response Event Fields", "breadcrumbs": "Network Requests & Console Message Capturing > Response Event Fields > Response Event Fields"}, {"id": "d599aa07299bc783", "url": "https://docs.crawl4ai.com/advanced/network-console-capture/", "page_title": "Network Requests & Console Message Capturing", "page_type": "guide", "page_summary": "This page explains how to capture network requests and browser console messages during a crawl using Crawl4AI, including configuration, example usage, data structures, benefits, and use cases.", "heading": "Failed Request Event Fields", "content": "Page: Network Requests & Console Message Capturing\nSection: Failed Request Event Fields\n\n```json\n{\n  \"event_type\": \"request_failed\",\n  \"url\": \"https://example.com/missing.png\",\n  \"method\": \"GET\",\n  \"resource_type\": \"image\",\n  \"failure_text\": \"net::ERR_ABORTED 404\",\n  \"timestamp\": 1633456789.789\n}\n```", "code_blocks": [{"language": "json", "code": "{\n  \"event_type\": \"request_failed\",\n  \"url\": \"https://example.com/missing.png\",\n  \"method\": \"GET\",\n  \"resource_type\": \"image\",\n  \"failure_text\": \"net::ERR_ABORTED 404\",\n  \"timestamp\": 1633456789.789\n}", "filename": ""}], "chunk_position": 12, "heading_path": "Failed Request Event Fields > Failed Request Event Fields", "breadcrumbs": "Network Requests & Console Message Capturing > Failed Request Event Fields > Failed Request Event Fields"}, {"id": "74a5cbdf1ffdb838", "url": "https://docs.crawl4ai.com/advanced/network-console-capture/", "page_title": "Network Requests & Console Message Capturing", "page_type": "guide", "page_summary": "This page explains how to capture network requests and browser console messages during a crawl using Crawl4AI, including configuration, example usage, data structures, benefits, and use cases.", "heading": "Console Messages", "content": "Page: Network Requests & Console Message Capturing\nSection: Console Messages\n\nThe `result.console_messages` contains a list of dictionaries, each representing a console message with these common fields:\n\n| Field | Description |\n| --- | --- |\n| `type` | Message type: `\"log\"`,…", "code_blocks": [], "chunk_position": 12, "heading_path": "Console Messages > Console Messages", "breadcrumbs": "Network Requests & Console Message Capturing > Console Messages > Console Messages"}, {"id": "4e4c8e20f58b8703", "url": "https://docs.crawl4ai.com/advanced/network-console-capture/", "page_title": "Network Requests & Console Message Capturing", "page_type": "guide", "page_summary": "This page explains how to capture network requests and browser console messages during a crawl using Crawl4AI, including configuration, example usage, data structures, benefits, and use cases.", "heading": "Console Message Example", "content": "Page: Network Requests & Console Message Capturing\nSection: Console Message Example\n\n```json\n{\n  \"type\": \"error\",\n  \"text\": \"Uncaught TypeError: Cannot read property 'length' of undefined\",\n  \"location\": \"https://example.com/script.js:123:45\",\n  \"timestamp\": 1633456790.123\n}\n```", "code_blocks": [{"language": "json", "code": "{\n  \"type\": \"error\",\n  \"text\": \"Uncaught TypeError: Cannot read property 'length' of undefined\",\n  \"location\": \"https://example.com/script.js:123:45\",\n  \"timestamp\": 1633456790.123\n}", "filename": ""}], "chunk_position": 12, "heading_path": "Console Message Example > Console Message Example", "breadcrumbs": "Network Requests & Console Message Capturing > Console Message Example > Console Message Example"}, {"id": "411eacea0c75ea5e", "url": "https://docs.crawl4ai.com/advanced/network-console-capture/", "page_title": "Network Requests & Console Message Capturing", "page_type": "guide", "page_summary": "This page explains how to capture network requests and browser console messages during a crawl using Crawl4AI, including configuration, example usage, data structures, benefits, and use cases.", "heading": "Key Benefits", "content": "Page: Network Requests & Console Message Capturing\nSection: Key Benefits\n\n**Full Request Visibility** : Capture all network activity including:\n- Requests (URLs, methods, headers, post data)\n- Responses (status codes, headers, timing)\n- Failed requests (with error…", "code_blocks": [], "chunk_position": 12, "heading_path": "Key Benefits > Key Benefits", "breadcrumbs": "Network Requests & Console Message Capturing > Key Benefits > Key Benefits"}, {"id": "db3935cab75fc93f", "url": "https://docs.crawl4ai.com/advanced/network-console-capture/", "page_title": "Network Requests & Console Message Capturing", "page_type": "guide", "page_summary": "This page explains how to capture network requests and browser console messages during a crawl using Crawl4AI, including configuration, example usage, data structures, benefits, and use cases.", "heading": "Use Cases", "content": "Page: Network Requests & Console Message Capturing\nSection: Use Cases\n\n**API Discovery** : Identify hidden endpoints and data flows in single-page applications\n\n**Debugging** : Track down JavaScript errors affecting page functionality\n\n**Security Auditing** : Detect…", "code_blocks": [], "chunk_position": 12, "heading_path": "Use Cases > Use Cases", "breadcrumbs": "Network Requests & Console Message Capturing > Use Cases > Use Cases"}, {"id": "663f346ee8d68baa", "url": "https://docs.crawl4ai.com/advanced/pdf-parsing/", "page_title": "PDF Processing Strategies", "page_type": "api", "page_summary": "This page describes the PDF processing strategies in Crawl4AI, including PDFCrawlerStrategy and PDFContentScrapingStrategy, which enable crawling and extracting content from PDF files.", "heading": "Overview", "content": "Page: PDF Processing Strategies\nSection: Overview\n\n`PDFCrawlerStrategy` is an implementation of `AsyncCrawlerStrategy` designed specifically for PDF documents. Instead of interpreting the input URL as an HTML webpage, this strategy treats it as a…", "code_blocks": [], "chunk_position": 13, "heading_path": "Overview > Overview", "breadcrumbs": "PDF Processing Strategies > Overview > Overview"}, {"id": "cf762e5161b4cc86", "url": "https://docs.crawl4ai.com/advanced/pdf-parsing/", "page_title": "PDF Processing Strategies", "page_type": "api", "page_summary": "This page describes the PDF processing strategies in Crawl4AI, including PDFCrawlerStrategy and PDFContentScrapingStrategy, which enable crawling and extracting content from PDF files.", "heading": "When to Use", "content": "Page: PDF Processing Strategies\nSection: When to Use\n\nUse `PDFCrawlerStrategy` when you need to:\n- Process PDF files using the `AsyncWebCrawler`.\n- Handle PDFs from both web URLs (e.g., `https://example.com/document.pdf`) and local file paths (e.g.,…", "code_blocks": [], "chunk_position": 13, "heading_path": "When to Use > When to Use", "breadcrumbs": "PDF Processing Strategies > When to Use > When to Use"}, {"id": "eeb50f40a0c1d79f", "url": "https://docs.crawl4ai.com/advanced/pdf-parsing/", "page_title": "PDF Processing Strategies", "page_type": "api", "page_summary": "This page describes the PDF processing strategies in Crawl4AI, including PDFCrawlerStrategy and PDFContentScrapingStrategy, which enable crawling and extracting content from PDF files.", "heading": "Key Methods and Their Behavior", "content": "Page: PDF Processing Strategies\nSection: Key Methods and Their Behavior\n\n- **`__init__(self, logger: AsyncLogger = None)`** :\n\n- Initializes the strategy.\n- `logger`: An optional `AsyncLogger` instance (from `crawl4ai.async_logger`) for logging purposes.\n\n- **`async…", "code_blocks": [], "chunk_position": 13, "heading_path": "Key Methods and Their Behavior > Key Methods and Their Behavior", "breadcrumbs": "PDF Processing Strategies > Key Methods and Their Behavior > Key Methods and Their Behavior"}, {"id": "6d5cd1bd8f68c568", "url": "https://docs.crawl4ai.com/advanced/pdf-parsing/", "page_title": "PDF Processing Strategies", "page_type": "api", "page_summary": "This page describes the PDF processing strategies in Crawl4AI, including PDFCrawlerStrategy and PDFContentScrapingStrategy, which enable crawling and extracting content from PDF files.", "heading": "Example Usage", "content": "Page: PDF Processing Strategies\nSection: Example Usage\n\n```python\nimport asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai.processors.pdf import PDFCrawlerStrategy, PDFContentScrapingStrategy\n\nasync def main():\n    # Initialize the PDF…\n```", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai.processors.pdf import PDFCrawlerStrategy, PDFContentScrapingStrategy\n\nasync def main():\n    # Initialize the PDF…", "filename": ""}], "chunk_position": 13, "heading_path": "Example Usage > Example Usage", "breadcrumbs": "PDF Processing Strategies > Example Usage > Example Usage"}, {"id": "a9889dff7017f458", "url": "https://docs.crawl4ai.com/advanced/pdf-parsing/", "page_title": "PDF Processing Strategies", "page_type": "api", "page_summary": "This page describes the PDF processing strategies in Crawl4AI, including PDFCrawlerStrategy and PDFContentScrapingStrategy, which enable crawling and extracting content from PDF files.", "heading": "Pros and Cons", "content": "Page: PDF Processing Strategies\nSection: Pros and Cons\n\n**Pros:** \n-   Enables `AsyncWebCrawler` to handle PDF sources directly using familiar `arun` calls.\n-   Provides a consistent interface for specifying PDF sources (URLs or local paths).\n-…", "code_blocks": [], "chunk_position": 13, "heading_path": "Pros and Cons > Pros and Cons", "breadcrumbs": "PDF Processing Strategies > Pros and Cons > Pros and Cons"}, {"id": "528753dc0efbaa13", "url": "https://docs.crawl4ai.com/advanced/pdf-parsing/", "page_title": "PDF Processing Strategies", "page_type": "api", "page_summary": "This page describes the PDF processing strategies in Crawl4AI, including PDFCrawlerStrategy and PDFContentScrapingStrategy, which enable crawling and extracting content from PDF files.", "heading": "Key Configuration Attributes", "content": "Page: PDF Processing Strategies\nSection: Key Configuration Attributes\n\nWhen initializing `PDFContentScrapingStrategy`, you can configure its behavior using the following attributes:\n-    **`extract_images: bool = False`** : If `True`, the strategy will attempt to…", "code_blocks": [], "chunk_position": 13, "heading_path": "Key Configuration Attributes > Key Configuration Attributes", "breadcrumbs": "PDF Processing Strategies > Key Configuration Attributes > Key Configuration Attributes"}, {"id": "eb84883856a55d0e", "url": "https://docs.crawl4ai.com/advanced/proxy-security/", "page_title": "Proxy & Security", "page_type": "guide", "page_summary": "This guide covers proxy configuration and security features in Crawl4AI, including SSL certificate analysis and proxy rotation strategies.", "heading": "Understanding Proxy Configuration", "content": "Page: Proxy & Security\nSection: Understanding Proxy Configuration\n\nCrawl4AI recommends configuring proxies per request through `CrawlerRunConfig.proxy_config`. This gives you precise control, enables rotation strategies, and keeps examples simple enough to copy,…", "code_blocks": [], "chunk_position": 14, "heading_path": "Understanding Proxy Configuration > Understanding Proxy Configuration", "breadcrumbs": "Proxy & Security > Understanding Proxy Configuration > Understanding Proxy Configuration"}, {"id": "c99022c5acc746a2", "url": "https://docs.crawl4ai.com/advanced/proxy-security/", "page_title": "Proxy & Security", "page_type": "guide", "page_summary": "This guide covers proxy configuration and security features in Crawl4AI, including SSL certificate analysis and proxy rotation strategies.", "heading": "Basic Proxy Setup", "content": "Page: Proxy & Security\nSection: Basic Proxy Setup\n\nConfigure proxies that apply to each crawl operation:\n\nWhy request-level?\n\n`CrawlerRunConfig.proxy_config` keeps each request self-contained, so swapping proxies or rotation strategies is just a…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, ProxyConfig\n\nrun_config = CrawlerRunConfig(proxy_config=ProxyConfig(server=\"http://proxy.example.com:8080\"))\n#…", "filename": ""}], "chunk_position": 14, "heading_path": "Basic Proxy Setup > Basic Proxy Setup", "breadcrumbs": "Proxy & Security > Basic Proxy Setup > Basic Proxy Setup"}, {"id": "064a85569150509b", "url": "https://docs.crawl4ai.com/advanced/proxy-security/", "page_title": "Proxy & Security", "page_type": "guide", "page_summary": "This guide covers proxy configuration and security features in Crawl4AI, including SSL certificate analysis and proxy rotation strategies.", "heading": "Supported Proxy Formats", "content": "Page: Proxy & Security\nSection: Supported Proxy Formats\n\nThe `ProxyConfig.from_string()` method supports multiple formats:", "code_blocks": [{"language": "python", "code": "from crawl4ai import ProxyConfig\n\n# HTTP proxy with authentication\nproxy1 = ProxyConfig.from_string(\"http://user:pass@192.168.1.1:8080\")\n\n# HTTPS proxy\nproxy2 =…", "filename": ""}], "chunk_position": 14, "heading_path": "Supported Proxy Formats > Supported Proxy Formats", "breadcrumbs": "Proxy & Security > Supported Proxy Formats > Supported Proxy Formats"}, {"id": "80a30389ae17d62b", "url": "https://docs.crawl4ai.com/advanced/proxy-security/", "page_title": "Proxy & Security", "page_type": "guide", "page_summary": "This guide covers proxy configuration and security features in Crawl4AI, including SSL certificate analysis and proxy rotation strategies.", "heading": "Environment Variable Configuration", "content": "Page: Proxy & Security\nSection: Environment Variable Configuration\n\nLoad proxies from environment variables for easy configuration:", "code_blocks": [{"language": "python", "code": "import os\nfrom crawl4ai import ProxyConfig, CrawlerRunConfig\n\n# Set environment variable\nos.environ[\"PROXIES\"] = \"ip1:port1:user1:pass1,ip2:port2:user2:pass2,ip3:port3\"\n\n# Load all proxies\nproxies =…", "filename": ""}], "chunk_position": 14, "heading_path": "Environment Variable Configuration > Environment Variable Configuration", "breadcrumbs": "Proxy & Security > Environment Variable Configuration > Environment Variable Configuration"}, {"id": "b790b8b46e8b49f3", "url": "https://docs.crawl4ai.com/advanced/proxy-security/", "page_title": "Proxy & Security", "page_type": "guide", "page_summary": "This guide covers proxy configuration and security features in Crawl4AI, including SSL certificate analysis and proxy rotation strategies.", "heading": "Rotating Proxies", "content": "Page: Proxy & Security\nSection: Rotating Proxies\n\nCrawl4AI supports automatic proxy rotation to distribute requests across multiple proxy servers. Rotation is applied per request using a rotation strategy on `CrawlerRunConfig`.", "code_blocks": [], "chunk_position": 14, "heading_path": "Rotating Proxies > Rotating Proxies", "breadcrumbs": "Proxy & Security > Rotating Proxies > Rotating Proxies"}, {"id": "9cd8e491ce7f0c48", "url": "https://docs.crawl4ai.com/advanced/proxy-security/", "page_title": "Proxy & Security", "page_type": "guide", "page_summary": "This guide covers proxy configuration and security features in Crawl4AI, including SSL certificate analysis and proxy rotation strategies.", "heading": "Proxy Rotation (recommended)", "content": "Page: Proxy & Security\nSection: Proxy Rotation (recommended)\n\n```python\nimport asyncio\nimport re\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode, ProxyConfig\nfrom crawl4ai.proxy_strategy import RoundRobinProxyStrategy\n\nasync def main():…\n```", "code_blocks": [{"language": "python", "code": "import asyncio\nimport re\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode, ProxyConfig\nfrom crawl4ai.proxy_strategy import RoundRobinProxyStrategy\n\nasync def main():…", "filename": ""}], "chunk_position": 14, "heading_path": "Proxy Rotation (recommended) > Proxy Rotation (recommended)", "breadcrumbs": "Proxy & Security > Proxy Rotation (recommended) > Proxy Rotation (recommended)"}, {"id": "647d40f39eaf06d3", "url": "https://docs.crawl4ai.com/advanced/proxy-security/", "page_title": "Proxy & Security", "page_type": "guide", "page_summary": "This guide covers proxy configuration and security features in Crawl4AI, including SSL certificate analysis and proxy rotation strategies.", "heading": "SSL Certificate Analysis", "content": "Page: Proxy & Security\nSection: SSL Certificate Analysis\n\nCombine proxy usage with SSL certificate inspection for enhanced security analysis. SSL certificate fetching is configured per request via `CrawlerRunConfig`.", "code_blocks": [], "chunk_position": 14, "heading_path": "SSL Certificate Analysis > SSL Certificate Analysis", "breadcrumbs": "Proxy & Security > SSL Certificate Analysis > SSL Certificate Analysis"}, {"id": "1426029e559ff565", "url": "https://docs.crawl4ai.com/advanced/proxy-security/", "page_title": "Proxy & Security", "page_type": "guide", "page_summary": "This guide covers proxy configuration and security features in Crawl4AI, including SSL certificate analysis and proxy rotation strategies.", "heading": "Per-Request SSL Certificate Analysis", "content": "Page: Proxy & Security\nSection: Per-Request SSL Certificate Analysis\n\n```python\nimport asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig\n\nrun_config = CrawlerRunConfig(\n    proxy_config={\n        \"server\": \"http://proxy.example.com:8080\",…\n```", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig\n\nrun_config = CrawlerRunConfig(\n    proxy_config={\n        \"server\": \"http://proxy.example.com:8080\",…", "filename": ""}], "chunk_position": 14, "heading_path": "Per-Request SSL Certificate Analysis > Per-Request SSL Certificate Analysis", "breadcrumbs": "Proxy & Security > Per-Request SSL Certificate Analysis > Per-Request SSL Certificate Analysis"}, {"id": "4fbb9ca28ca8f003", "url": "https://docs.crawl4ai.com/advanced/proxy-security/", "page_title": "Proxy & Security", "page_type": "guide", "page_summary": "This guide covers proxy configuration and security features in Crawl4AI, including SSL certificate analysis and proxy rotation strategies.", "heading": "1. Proxy Rotation for Anonymity", "content": "Page: Proxy & Security\nSection: 1. Proxy Rotation for Anonymity\n\n```python\nfrom crawl4ai import CrawlerRunConfig, ProxyConfig\nfrom crawl4ai.proxy_strategy import RoundRobinProxyStrategy\n\n# Use multiple proxies to avoid IP blocking\nproxies =…\n```", "code_blocks": [{"language": "python", "code": "from crawl4ai import CrawlerRunConfig, ProxyConfig\nfrom crawl4ai.proxy_strategy import RoundRobinProxyStrategy\n\n# Use multiple proxies to avoid IP blocking\nproxies =…", "filename": ""}], "chunk_position": 14, "heading_path": "1. Proxy Rotation for Anonymity > 1. Proxy Rotation for Anonymity", "breadcrumbs": "Proxy & Security > 1. Proxy Rotation for Anonymity > 1. Proxy Rotation for Anonymity"}, {"id": "677d2343d0a4b362", "url": "https://docs.crawl4ai.com/advanced/proxy-security/", "page_title": "Proxy & Security", "page_type": "guide", "page_summary": "This guide covers proxy configuration and security features in Crawl4AI, including SSL certificate analysis and proxy rotation strategies.", "heading": "2. SSL Certificate Verification", "content": "Page: Proxy & Security\nSection: 2. SSL Certificate Verification\n\n```python\nfrom crawl4ai import CrawlerRunConfig\n\n# Always verify SSL certificates when possible\n# Per-request (affects specific requests)\nrun_config = CrawlerRunConfig(fetch_ssl_certificate=True)\n```", "code_blocks": [{"language": "python", "code": "from crawl4ai import CrawlerRunConfig\n\n# Always verify SSL certificates when possible\n# Per-request (affects specific requests)\nrun_config = CrawlerRunConfig(fetch_ssl_certificate=True)", "filename": ""}], "chunk_position": 14, "heading_path": "2. SSL Certificate Verification > 2. SSL Certificate Verification", "breadcrumbs": "Proxy & Security > 2. SSL Certificate Verification > 2. SSL Certificate Verification"}, {"id": "d431e5ec09bfc098", "url": "https://docs.crawl4ai.com/advanced/proxy-security/", "page_title": "Proxy & Security", "page_type": "guide", "page_summary": "This guide covers proxy configuration and security features in Crawl4AI, including SSL certificate analysis and proxy rotation strategies.", "heading": "3. Environment Variable Security", "content": "Page: Proxy & Security\nSection: 3. Environment Variable Security\n\n```bash\n# Use environment variables for sensitive proxy credentials\n# Avoid hardcoding usernames/passwords in code\nexport PROXIES=\"ip1:port1:user1:pass1,ip2:port2:user2:pass2\"\n```", "code_blocks": [{"language": "bash", "code": "# Use environment variables for sensitive proxy credentials\n# Avoid hardcoding usernames/passwords in code\nexport PROXIES=\"ip1:port1:user1:pass1,ip2:port2:user2:pass2\"", "filename": ""}], "chunk_position": 14, "heading_path": "3. Environment Variable Security > 3. Environment Variable Security", "breadcrumbs": "Proxy & Security > 3. Environment Variable Security > 3. Environment Variable Security"}, {"id": "478e23ab1c5398c9", "url": "https://docs.crawl4ai.com/advanced/proxy-security/", "page_title": "Proxy & Security", "page_type": "guide", "page_summary": "This guide covers proxy configuration and security features in Crawl4AI, including SSL certificate analysis and proxy rotation strategies.", "heading": "4. SOCKS5 for Enhanced Security", "content": "Page: Proxy & Security\nSection: 4. SOCKS5 for Enhanced Security\n\n```python\nfrom crawl4ai import CrawlerRunConfig\n\n# Prefer SOCKS5 proxies for better protocol support\nrun_config = CrawlerRunConfig(proxy_config=\"socks5://proxy.example.com:1080\")\n```", "code_blocks": [{"language": "python", "code": "from crawl4ai import CrawlerRunConfig\n\n# Prefer SOCKS5 proxies for better protocol support\nrun_config = CrawlerRunConfig(proxy_config=\"socks5://proxy.example.com:1080\")", "filename": ""}], "chunk_position": 14, "heading_path": "4. SOCKS5 for Enhanced Security > 4. SOCKS5 for Enhanced Security", "breadcrumbs": "Proxy & Security > 4. SOCKS5 for Enhanced Security > 4. SOCKS5 for Enhanced Security"}, {"id": "50f028f9f956de24", "url": "https://docs.crawl4ai.com/advanced/proxy-security/", "page_title": "Proxy & Security", "page_type": "guide", "page_summary": "This guide covers proxy configuration and security features in Crawl4AI, including SSL certificate analysis and proxy rotation strategies.", "heading": "Migration from Deprecated `proxy` Parameter", "content": "Page: Proxy & Security\nSection: Migration from Deprecated `proxy` Parameter\n\nThe legacy `proxy` argument on `BrowserConfig` is deprecated. Configure proxies through `CrawlerRunConfig.proxy_config` so each request fully describes its network settings.", "code_blocks": [{"language": "python", "code": "# Old (deprecated) approach\n# from crawl4ai import BrowserConfig\n# browser_config = BrowserConfig(proxy_config=\"http://proxy.example.com:8080\")\n\n# New (preferred) approach\nfrom crawl4ai import…", "filename": ""}], "chunk_position": 14, "heading_path": "Migration from Deprecated `proxy` Parameter > Migration from Deprecated `proxy` Parameter", "breadcrumbs": "Proxy & Security > Migration from Deprecated `proxy` Parameter > Migration from Deprecated `proxy` Parameter"}, {"id": "a8753928ad301c84", "url": "https://docs.crawl4ai.com/advanced/proxy-security/", "page_title": "Proxy & Security", "page_type": "guide", "page_summary": "This guide covers proxy configuration and security features in Crawl4AI, including SSL certificate analysis and proxy rotation strategies.", "heading": "Safe Logging of Proxies", "content": "Page: Proxy & Security\nSection: Safe Logging of Proxies\n\n```python\nfrom crawl4ai import ProxyConfig\n\ndef safe_proxy_repr(proxy: ProxyConfig):\n    if getattr(proxy, \"username\", None):\n        return f\"{proxy.server} (auth: ****)\"\n    return proxy.server\n```", "code_blocks": [{"language": "python", "code": "from crawl4ai import ProxyConfig\n\ndef safe_proxy_repr(proxy: ProxyConfig):\n    if getattr(proxy, \"username\", None):\n        return f\"{proxy.server} (auth: ****)\"\n    return proxy.server", "filename": ""}], "chunk_position": 14, "heading_path": "Safe Logging of Proxies > Safe Logging of Proxies", "breadcrumbs": "Proxy & Security > Safe Logging of Proxies > Safe Logging of Proxies"}, {"id": "760bf8c9eff9d23b", "url": "https://docs.crawl4ai.com/advanced/proxy-security/", "page_title": "Proxy & Security", "page_type": "guide", "page_summary": "This guide covers proxy configuration and security features in Crawl4AI, including SSL certificate analysis and proxy rotation strategies.", "heading": "Common Issues", "content": "Page: Proxy & Security\nSection: Common Issues\n\nProxy connection failed\n\n- Verify the proxy server is reachable from your network.\n- Double-check authentication credentials.\n- Ensure the protocol matches (`http`, `https`, or `socks5`).\n\nSSL…", "code_blocks": [], "chunk_position": 14, "heading_path": "Common Issues > Common Issues", "breadcrumbs": "Proxy & Security > Common Issues > Common Issues"}, {"id": "3e5cd38af4172e17", "url": "https://docs.crawl4ai.com/advanced/proxy-security/", "page_title": "Proxy & Security", "page_type": "guide", "page_summary": "This guide covers proxy configuration and security features in Crawl4AI, including SSL certificate analysis and proxy rotation strategies.", "heading": "See Also", "content": "Page: Proxy & Security\nSection: See Also\n\n[Anti-Bot Detection & Fallback](../anti-bot-and-fallback/) — Automatic retry with proxy escalation and fallback functions when anti-bot blocking is detected", "code_blocks": [], "chunk_position": 14, "heading_path": "See Also > See Also", "breadcrumbs": "Proxy & Security > See Also > See Also"}, {"id": "a30a2f3ee07cf062", "url": "https://docs.crawl4ai.com/advanced/session-management/", "page_title": "Session Management - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains how to use session management in Crawl4AI to maintain state across multiple requests, enabling sequential crawling, dynamic content handling, and advanced techniques like custom…", "heading": "Overview", "content": "Page: Session Management - Crawl4AI Documentation (v0.9.x)\nSection: Overview\n\nSession management in Crawl4AI is a powerful feature that allows you to maintain state across multiple requests, making it particularly suitable for handling complex multi-step crawling tasks. It…", "code_blocks": [], "chunk_position": 15, "heading_path": "Overview > Overview", "breadcrumbs": "Session Management - Crawl4AI Documentation (v0.9.x) > Overview > Overview"}, {"id": "25452d92b5201e21", "url": "https://docs.crawl4ai.com/advanced/session-management/", "page_title": "Session Management - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains how to use session management in Crawl4AI to maintain state across multiple requests, enabling sequential crawling, dynamic content handling, and advanced techniques like custom…", "heading": "Basic Session Usage", "content": "Page: Session Management - Crawl4AI Documentation (v0.9.x)\nSection: Basic Session Usage\n\nUse `BrowserConfig` and `CrawlerRunConfig` to maintain state with a `session_id`:", "code_blocks": [{"language": "python", "code": "from crawl4ai.async_configs import BrowserConfig, CrawlerRunConfig\n\nasync with AsyncWebCrawler() as crawler:\n    session_id = \"my_session\"\n\n    # Define configurations\n    config1 =…", "filename": ""}], "chunk_position": 15, "heading_path": "Basic Session Usage > Basic Session Usage", "breadcrumbs": "Session Management - Crawl4AI Documentation (v0.9.x) > Basic Session Usage > Basic Session Usage"}, {"id": "609c440a598cf257", "url": "https://docs.crawl4ai.com/advanced/session-management/", "page_title": "Session Management - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains how to use session management in Crawl4AI to maintain state across multiple requests, enabling sequential crawling, dynamic content handling, and advanced techniques like custom…", "heading": "Dynamic Content with Sessions", "content": "Page: Session Management - Crawl4AI Documentation (v0.9.x)\nSection: Dynamic Content with Sessions\n\nHere's an example of crawling GitHub commits across multiple pages while preserving session state:", "code_blocks": [{"language": "python", "code": "from crawl4ai.async_configs import CrawlerRunConfig\nfrom crawl4ai import JsonCssExtractionStrategy\nfrom crawl4ai.cache_context import CacheMode\n\nasync def crawl_dynamic_content():\n    url =…", "filename": ""}], "chunk_position": 15, "heading_path": "Dynamic Content with Sessions > Dynamic Content with Sessions", "breadcrumbs": "Session Management - Crawl4AI Documentation (v0.9.x) > Dynamic Content with Sessions > Dynamic Content with Sessions"}, {"id": "65d297e84a1d1b20", "url": "https://docs.crawl4ai.com/advanced/session-management/", "page_title": "Session Management - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains how to use session management in Crawl4AI to maintain state across multiple requests, enabling sequential crawling, dynamic content handling, and advanced techniques like custom…", "heading": "Example 1: Basic Session-Based Crawling", "content": "Page: Session Management - Crawl4AI Documentation (v0.9.x)\nSection: Example 1: Basic Session-Based Crawling\n\nA simple example using session-based crawling:\n\nThis example shows:\n1. Reusing the same `session_id` across multiple requests.\n2. Executing JavaScript to load more content dynamically.\n3. Properly…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai.async_configs import BrowserConfig, CrawlerRunConfig\nfrom crawl4ai.cache_context import CacheMode\n\nasync def basic_session_crawl():\n    async with AsyncWebCrawler() as…", "filename": ""}], "chunk_position": 15, "heading_path": "Example 1: Basic Session-Based Crawling > Example 1: Basic Session-Based Crawling", "breadcrumbs": "Session Management - Crawl4AI Documentation (v0.9.x) > Example 1: Basic Session-Based Crawling > Example 1: Basic Session-Based Crawling"}, {"id": "9697acd946a9864c", "url": "https://docs.crawl4ai.com/advanced/session-management/", "page_title": "Session Management - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains how to use session management in Crawl4AI to maintain state across multiple requests, enabling sequential crawling, dynamic content handling, and advanced techniques like custom…", "heading": "Advanced Technique 1: Custom Execution Hooks", "content": "Page: Session Management - Crawl4AI Documentation (v0.9.x)\nSection: Advanced Technique 1: Custom Execution Hooks\n\n> Warning: You might feel confused by the end of the next few examples 😅, so make sure you are comfortable with the order of the parts before you start this.\n\nUse custom hooks to handle complex…", "code_blocks": [{"language": "python", "code": "async def advanced_session_crawl_with_hooks():\n    first_commit = \"\"\n\n    async def on_execution_started(page):\n        nonlocal first_commit\n        try:\n            while True:…", "filename": ""}], "chunk_position": 15, "heading_path": "Advanced Technique 1: Custom Execution Hooks > Advanced Technique 1: Custom Execution Hooks", "breadcrumbs": "Session Management - Crawl4AI Documentation (v0.9.x) > Advanced Technique 1: Custom Execution Hooks > Advanced Technique 1: Custom Execution Hooks"}, {"id": "b5209176750032c8", "url": "https://docs.crawl4ai.com/advanced/session-management/", "page_title": "Session Management - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains how to use session management in Crawl4AI to maintain state across multiple requests, enabling sequential crawling, dynamic content handling, and advanced techniques like custom…", "heading": "Advanced Technique 2: Integrated JavaScript Execution and Waiting", "content": "Page: Session Management - Crawl4AI Documentation (v0.9.x)\nSection: Advanced Technique 2: Integrated JavaScript Execution and Waiting\n\nCombine JavaScript execution and waiting logic for concise handling of dynamic content:", "code_blocks": [{"language": "python", "code": "async def integrated_js_and_wait_crawl():\n    async with AsyncWebCrawler() as crawler:\n        session_id = \"integrated_session\"\n        url = \"https://github.com/example/repo/commits/main\"…", "filename": ""}], "chunk_position": 15, "heading_path": "Advanced Technique 2: Integrated JavaScript Execution and Waiting > Advanced Technique 2: Integrated JavaScript Execution and Waiting", "breadcrumbs": "Session Management - Crawl4AI Documentation (v0.9.x) > Advanced Technique 2: Integrated JavaScript Execution and Waiting > Advanced Technique 2: Integrated JavaScript Execution and Waiting"}, {"id": "71d95c13f60a68a1", "url": "https://docs.crawl4ai.com/advanced/session-management/", "page_title": "Session Management - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains how to use session management in Crawl4AI to maintain state across multiple requests, enabling sequential crawling, dynamic content handling, and advanced techniques like custom…", "heading": "Common Use Cases for Sessions", "content": "Page: Session Management - Crawl4AI Documentation (v0.9.x)\nSection: Common Use Cases for Sessions\n\n1. **Authentication Flows**: Login and interact with secured pages.\n\n2. **Pagination Handling**: Navigate through multiple pages.\n\n3. **Form Submissions**: Fill forms, submit, and process…", "code_blocks": [], "chunk_position": 15, "heading_path": "Common Use Cases for Sessions > Common Use Cases for Sessions", "breadcrumbs": "Session Management - Crawl4AI Documentation (v0.9.x) > Common Use Cases for Sessions > Common Use Cases for Sessions"}, {"id": "16a22fd81574de26", "url": "https://docs.crawl4ai.com/advanced/ssl-certificate/", "page_title": "SSLCertificate Reference", "page_type": "api", "page_summary": "Reference for the SSLCertificate class in Crawl4AI, covering how to load, inspect, and export SSL/TLS certificate data, and how to use it with fetch_ssl_certificate=True in CrawlerRunConfig.", "heading": "1. Overview", "content": "Page: SSLCertificate Reference\nSection: 1. Overview\n\nThe **`SSLCertificate`** class encapsulates an SSL certificate’s data and allows exporting it in various formats (PEM, DER, JSON, or text). It’s used within **Crawl4AI** whenever you set…", "code_blocks": [{"language": "python", "code": "class SSLCertificate:\n    \"\"\"\n    Represents an SSL certificate with methods to export in various formats.\n\n    Main Methods:\n    - from_url(url, timeout=10)\n    - from_file(file_path)\n    -…", "filename": ""}], "chunk_position": 16, "heading_path": "1. Overview > 1. Overview", "breadcrumbs": "SSLCertificate Reference > 1. Overview > 1. Overview"}, {"id": "071725c1e697ff12", "url": "https://docs.crawl4ai.com/advanced/ssl-certificate/", "page_title": "SSLCertificate Reference", "page_type": "api", "page_summary": "Reference for the SSLCertificate class in Crawl4AI, covering how to load, inspect, and export SSL/TLS certificate data, and how to use it with fetch_ssl_certificate=True in CrawlerRunConfig.", "heading": "Typical Use Case", "content": "Page: SSLCertificate Reference\nSection: Typical Use Case\n\n- You **enable** certificate fetching in your crawl by:\n- After `arun()`, if `result.ssl_certificate` is present, it’s an instance of **`SSLCertificate`**.\n- You can **read** basic properties…", "code_blocks": [{"language": "python", "code": "CrawlerRunConfig(fetch_ssl_certificate=True, ...)", "filename": ""}], "chunk_position": 16, "heading_path": "Typical Use Case > Typical Use Case", "breadcrumbs": "SSLCertificate Reference > Typical Use Case > Typical Use Case"}, {"id": "fe7c2ad78bb7fc30", "url": "https://docs.crawl4ai.com/advanced/ssl-certificate/", "page_title": "SSLCertificate Reference", "page_type": "api", "page_summary": "Reference for the SSLCertificate class in Crawl4AI, covering how to load, inspect, and export SSL/TLS certificate data, and how to use it with fetch_ssl_certificate=True in CrawlerRunConfig.", "heading": "2.1 from_url(url, timeout=10)", "content": "Page: SSLCertificate Reference\nSection: 2.1 from_url(url, timeout=10)\n\nManually load an SSL certificate from a given URL (port 443). Typically used internally, but you can call it directly if you want:", "code_blocks": [{"language": "python", "code": "cert = SSLCertificate.from_url(\"https://example.com\")\nif cert:\n    print(\"Fingerprint:\", cert.fingerprint)", "filename": ""}], "chunk_position": 16, "heading_path": "2.1 from_url(url, timeout=10) > 2.1 from_url(url, timeout=10)", "breadcrumbs": "SSLCertificate Reference > 2.1 from_url(url, timeout=10) > 2.1 from_url(url, timeout=10)"}, {"id": "cf320103ba576956", "url": "https://docs.crawl4ai.com/advanced/ssl-certificate/", "page_title": "SSLCertificate Reference", "page_type": "api", "page_summary": "Reference for the SSLCertificate class in Crawl4AI, covering how to load, inspect, and export SSL/TLS certificate data, and how to use it with fetch_ssl_certificate=True in CrawlerRunConfig.", "heading": "2.2 from_file(file_path)", "content": "Page: SSLCertificate Reference\nSection: 2.2 from_file(file_path)\n\nLoad from a file containing certificate data in ASN.1 or DER. Rarely needed unless you have local cert files:", "code_blocks": [{"language": "python", "code": "cert = SSLCertificate.from_file(\"/path/to/cert.der\")", "filename": ""}], "chunk_position": 16, "heading_path": "2.2 from_file(file_path) > 2.2 from_file(file_path)", "breadcrumbs": "SSLCertificate Reference > 2.2 from_file(file_path) > 2.2 from_file(file_path)"}, {"id": "1ff7f6f3767a7278", "url": "https://docs.crawl4ai.com/advanced/ssl-certificate/", "page_title": "SSLCertificate Reference", "page_type": "api", "page_summary": "Reference for the SSLCertificate class in Crawl4AI, covering how to load, inspect, and export SSL/TLS certificate data, and how to use it with fetch_ssl_certificate=True in CrawlerRunConfig.", "heading": "2.3 from_binary(binary_data)", "content": "Page: SSLCertificate Reference\nSection: 2.3 from_binary(binary_data)\n\nInitialize from raw binary. E.g., if you captured it from a socket or another source:", "code_blocks": [{"language": "python", "code": "cert = SSLCertificate.from_binary(raw_bytes)", "filename": ""}], "chunk_position": 16, "heading_path": "2.3 from_binary(binary_data) > 2.3 from_binary(binary_data)", "breadcrumbs": "SSLCertificate Reference > 2.3 from_binary(binary_data) > 2.3 from_binary(binary_data)"}, {"id": "5c39fb7aa169ea85", "url": "https://docs.crawl4ai.com/advanced/ssl-certificate/", "page_title": "SSLCertificate Reference", "page_type": "api", "page_summary": "Reference for the SSLCertificate class in Crawl4AI, covering how to load, inspect, and export SSL/TLS certificate data, and how to use it with fetch_ssl_certificate=True in CrawlerRunConfig.", "heading": "3. Common Properties", "content": "Page: SSLCertificate Reference\nSection: 3. Common Properties\n\nAfter obtaining a **`SSLCertificate`** instance (e.g. `result.ssl_certificate` from a crawl), you can read:\n\n1. **`issuer`** *(dict)*\n   - E.g. `{\"CN\": \"My Root CA\", \"O\": \"...\"}`\n2. **`subject`**…", "code_blocks": [], "chunk_position": 16, "heading_path": "3. Common Properties > 3. Common Properties", "breadcrumbs": "SSLCertificate Reference > 3. Common Properties > 3. Common Properties"}, {"id": "d16a9d339ddaae56", "url": "https://docs.crawl4ai.com/advanced/ssl-certificate/", "page_title": "SSLCertificate Reference", "page_type": "api", "page_summary": "Reference for the SSLCertificate class in Crawl4AI, covering how to load, inspect, and export SSL/TLS certificate data, and how to use it with fetch_ssl_certificate=True in CrawlerRunConfig.", "heading": "4. Export Methods", "content": "Page: SSLCertificate Reference\nSection: 4. Export Methods\n\nOnce you have a **`SSLCertificate`** object, you can **export** or **inspect** it:", "code_blocks": [], "chunk_position": 16, "heading_path": "4. Export Methods > 4. Export Methods", "breadcrumbs": "SSLCertificate Reference > 4. Export Methods > 4. Export Methods"}, {"id": "9e2c8409f857742d", "url": "https://docs.crawl4ai.com/advanced/ssl-certificate/", "page_title": "SSLCertificate Reference", "page_type": "api", "page_summary": "Reference for the SSLCertificate class in Crawl4AI, covering how to load, inspect, and export SSL/TLS certificate data, and how to use it with fetch_ssl_certificate=True in CrawlerRunConfig.", "heading": "4.1 to_json(filepath=None) → Optional[str]", "content": "Page: SSLCertificate Reference\nSection: 4.1 to_json(filepath=None) → Optional[str]\n\n- Returns a JSON string containing the parsed certificate fields.\n- If `filepath` is provided, saves it to disk instead, returning `None`.\n\n**Usage**:", "code_blocks": [{"language": "python", "code": "json_data = cert.to_json()  # returns JSON string\ncert.to_json(\"certificate.json\")  # writes file, returns None", "filename": ""}], "chunk_position": 16, "heading_path": "4.1 to_json(filepath=None) → Optional[str] > 4.1 to_json(filepath=None) → Optional[str]", "breadcrumbs": "SSLCertificate Reference > 4.1 to_json(filepath=None) → Optional[str] > 4.1 to_json(filepath=None) → Optional[str]"}, {"id": "92f14b086371c40b", "url": "https://docs.crawl4ai.com/advanced/ssl-certificate/", "page_title": "SSLCertificate Reference", "page_type": "api", "page_summary": "Reference for the SSLCertificate class in Crawl4AI, covering how to load, inspect, and export SSL/TLS certificate data, and how to use it with fetch_ssl_certificate=True in CrawlerRunConfig.", "heading": "4.2 to_pem(filepath=None) → Optional[str]", "content": "Page: SSLCertificate Reference\nSection: 4.2 to_pem(filepath=None) → Optional[str]\n\n- Returns a PEM-encoded string (common for web servers).\n- If `filepath` is provided, saves it to disk instead.", "code_blocks": [{"language": "python", "code": "pem_str = cert.to_pem()              # in-memory PEM string\ncert.to_pem(\"/path/to/cert.pem\")     # saved to file", "filename": ""}], "chunk_position": 16, "heading_path": "4.2 to_pem(filepath=None) → Optional[str] > 4.2 to_pem(filepath=None) → Optional[str]", "breadcrumbs": "SSLCertificate Reference > 4.2 to_pem(filepath=None) → Optional[str] > 4.2 to_pem(filepath=None) → Optional[str]"}, {"id": "1b41f13ce088cbc7", "url": "https://docs.crawl4ai.com/advanced/ssl-certificate/", "page_title": "SSLCertificate Reference", "page_type": "api", "page_summary": "Reference for the SSLCertificate class in Crawl4AI, covering how to load, inspect, and export SSL/TLS certificate data, and how to use it with fetch_ssl_certificate=True in CrawlerRunConfig.", "heading": "4.3 to_der(filepath=None) → Optional[bytes]", "content": "Page: SSLCertificate Reference\nSection: 4.3 to_der(filepath=None) → Optional[bytes]\n\n- Returns the original DER (binary ASN.1) bytes.\n- If `filepath` is specified, writes the bytes there instead.", "code_blocks": [{"language": "python", "code": "der_bytes = cert.to_der()\ncert.to_der(\"certificate.der\")", "filename": ""}], "chunk_position": 16, "heading_path": "4.3 to_der(filepath=None) → Optional[bytes] > 4.3 to_der(filepath=None) → Optional[bytes]", "breadcrumbs": "SSLCertificate Reference > 4.3 to_der(filepath=None) → Optional[bytes] > 4.3 to_der(filepath=None) → Optional[bytes]"}, {"id": "8e5aa5e9a96385e8", "url": "https://docs.crawl4ai.com/advanced/ssl-certificate/", "page_title": "SSLCertificate Reference", "page_type": "api", "page_summary": "Reference for the SSLCertificate class in Crawl4AI, covering how to load, inspect, and export SSL/TLS certificate data, and how to use it with fetch_ssl_certificate=True in CrawlerRunConfig.", "heading": "4.4 (Optional) export_as_text()", "content": "Page: SSLCertificate Reference\nSection: 4.4 (Optional) export_as_text()\n\n- If you see a method like `export_as_text()`, it typically returns an OpenSSL-style textual representation.\n- Not always needed, but can help for debugging or manual inspection.", "code_blocks": [], "chunk_position": 16, "heading_path": "4.4 (Optional) export_as_text() > 4.4 (Optional) export_as_text()", "breadcrumbs": "SSLCertificate Reference > 4.4 (Optional) export_as_text() > 4.4 (Optional) export_as_text()"}, {"id": "6c57d250b10c1b12", "url": "https://docs.crawl4ai.com/advanced/ssl-certificate/", "page_title": "SSLCertificate Reference", "page_type": "api", "page_summary": "Reference for the SSLCertificate class in Crawl4AI, covering how to load, inspect, and export SSL/TLS certificate data, and how to use it with fetch_ssl_certificate=True in CrawlerRunConfig.", "heading": "5. Example Usage in Crawl4AI", "content": "Page: SSLCertificate Reference\nSection: 5. Example Usage in Crawl4AI\n\nBelow is a minimal sample showing how the crawler obtains an SSL cert from a site, then reads or exports it. The code snippet:", "code_blocks": [{"language": "python", "code": "import asyncio\nimport os\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\n\nasync def main():\n    tmp_dir = \"tmp\"\n    os.makedirs(tmp_dir, exist_ok=True)\n\n    config =…", "filename": ""}], "chunk_position": 16, "heading_path": "5. Example Usage in Crawl4AI > 5. Example Usage in Crawl4AI", "breadcrumbs": "SSLCertificate Reference > 5. Example Usage in Crawl4AI > 5. Example Usage in Crawl4AI"}, {"id": "9375778eecda3ea2", "url": "https://docs.crawl4ai.com/advanced/ssl-certificate/", "page_title": "SSLCertificate Reference", "page_type": "api", "page_summary": "Reference for the SSLCertificate class in Crawl4AI, covering how to load, inspect, and export SSL/TLS certificate data, and how to use it with fetch_ssl_certificate=True in CrawlerRunConfig.", "heading": "6. Notes & Best Practices", "content": "Page: SSLCertificate Reference\nSection: 6. Notes & Best Practices\n\n1. **Timeout**: `SSLCertificate.from_url` internally uses a default **10s** socket connect and wraps SSL.\n2. **Binary Form**: The certificate is loaded in ASN.1 (DER) form, then re-parsed by…", "code_blocks": [], "chunk_position": 16, "heading_path": "6. Notes & Best Practices > 6. Notes & Best Practices", "breadcrumbs": "SSLCertificate Reference > 6. Notes & Best Practices > 6. Notes & Best Practices"}, {"id": "1b45614a3a6c2fbb", "url": "https://docs.crawl4ai.com/advanced/ssl-certificate/", "page_title": "SSLCertificate Reference", "page_type": "api", "page_summary": "Reference for the SSLCertificate class in Crawl4AI, covering how to load, inspect, and export SSL/TLS certificate data, and how to use it with fetch_ssl_certificate=True in CrawlerRunConfig.", "heading": "Summary", "content": "Page: SSLCertificate Reference\nSection: Summary\n\n- **`SSLCertificate`** is a convenience class for capturing and exporting the **TLS certificate** from your crawled site(s).\n- Common usage is in the **`CrawlResult.ssl_certificate`** field,…", "code_blocks": [], "chunk_position": 16, "heading_path": "Summary > Summary", "breadcrumbs": "SSLCertificate Reference > Summary > Summary"}, {"id": "7e4e83616d371585", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Overview", "content": "Page: Undetected Browser Mode\nSection: Overview\n\nCrawl4AI offers two powerful anti-bot features to help you access websites with bot detection:\n\n- **Stealth Mode** - Uses playwright-stealth to modify browser fingerprints and behaviors\n-…", "code_blocks": [], "chunk_position": 17, "heading_path": "Overview > Overview", "breadcrumbs": "Undetected Browser Mode > Overview > Overview"}, {"id": "6de7b88b8a323191", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Anti-Bot Features Comparison", "content": "Page: Undetected Browser Mode\nSection: Anti-Bot Features Comparison\n\n| Feature | Regular Browser | Stealth Mode | Undetected Browser |\n| --- | --- | --- | --- |\n| WebDriver Detection | ❌ | ✅ | ✅ |\n| Navigator Properties | ❌ | ✅ | ✅ |\n| Plugin Emulation | ❌ | ✅ | ✅ |\n|…", "code_blocks": [], "chunk_position": 17, "heading_path": "Anti-Bot Features Comparison > Anti-Bot Features Comparison", "breadcrumbs": "Undetected Browser Mode > Anti-Bot Features Comparison > Anti-Bot Features Comparison"}, {"id": "6d246bef2b3d9df5", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Use Regular Browser + Stealth Mode When:", "content": "Page: Undetected Browser Mode\nSection: Use Regular Browser + Stealth Mode When:\n\n- Sites have basic bot detection (checking navigator.webdriver, plugins, etc.)\n- You need good performance with basic protection\n- Sites check for common automation indicators", "code_blocks": [], "chunk_position": 17, "heading_path": "Use Regular Browser + Stealth Mode When: > Use Regular Browser + Stealth Mode When:", "breadcrumbs": "Undetected Browser Mode > Use Regular Browser + Stealth Mode When: > Use Regular Browser + Stealth Mode When:"}, {"id": "adc71e1ea79bd9e1", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Use Undetected Browser When:", "content": "Page: Undetected Browser Mode\nSection: Use Undetected Browser When:\n\n- Sites employ sophisticated bot detection services (Cloudflare, DataDome, etc.)\n- Stealth mode alone isn't sufficient\n- You're willing to trade some performance for better evasion", "code_blocks": [], "chunk_position": 17, "heading_path": "Use Undetected Browser When: > Use Undetected Browser When:", "breadcrumbs": "Undetected Browser Mode > Use Undetected Browser When: > Use Undetected Browser When:"}, {"id": "04f5a95c520f1087", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Best Practice: Progressive Enhancement", "content": "Page: Undetected Browser Mode\nSection: Best Practice: Progressive Enhancement\n\n- **Start with**: Regular browser + Stealth mode\n- **If blocked**: Switch to Undetected browser\n- **If still blocked**: Combine Undetected browser + Stealth mode", "code_blocks": [], "chunk_position": 17, "heading_path": "Best Practice: Progressive Enhancement > Best Practice: Progressive Enhancement", "breadcrumbs": "Undetected Browser Mode > Best Practice: Progressive Enhancement > Best Practice: Progressive Enhancement"}, {"id": "058136b42fde8173", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Stealth Mode", "content": "Page: Undetected Browser Mode\nSection: Stealth Mode\n\nStealth mode is the simpler anti-bot solution that works with both regular and undetected browsers:", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler, BrowserConfig\n\n# Enable stealth mode with regular browser\nbrowser_config = BrowserConfig(\n    enable_stealth=True,  # Simple flag to enable\n    headless=False…", "filename": ""}], "chunk_position": 17, "heading_path": "Stealth Mode > Stealth Mode", "breadcrumbs": "Undetected Browser Mode > Stealth Mode > Stealth Mode"}, {"id": "0edc831659607cee", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "What Stealth Mode Does:", "content": "Page: Undetected Browser Mode\nSection: What Stealth Mode Does:\n\n- Removes `navigator.webdriver` flag\n- Modifies browser fingerprints\n- Emulates realistic plugin behavior\n- Adjusts navigator properties\n- Fixes common automation leaks", "code_blocks": [], "chunk_position": 17, "heading_path": "What Stealth Mode Does: > What Stealth Mode Does:", "breadcrumbs": "Undetected Browser Mode > What Stealth Mode Does: > What Stealth Mode Does:"}, {"id": "6c853cae369509ef", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Undetected Browser Mode", "content": "Page: Undetected Browser Mode\nSection: Undetected Browser Mode\n\nFor sites with sophisticated bot detection that stealth mode can't bypass, use the undetected browser adapter:", "code_blocks": [], "chunk_position": 17, "heading_path": "Undetected Browser Mode > Undetected Browser Mode", "breadcrumbs": "Undetected Browser Mode > Undetected Browser Mode > Undetected Browser Mode"}, {"id": "91f4954bf8f9ed6d", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Key Features", "content": "Page: Undetected Browser Mode\nSection: Key Features\n\n- **Drop-in Replacement**: Uses the same API as regular browser mode\n- **Enhanced Stealth**: Built-in patches to evade common detection methods\n- **Browser Adapter Pattern**: Seamlessly switch…", "code_blocks": [], "chunk_position": 17, "heading_path": "Key Features > Key Features", "breadcrumbs": "Undetected Browser Mode > Key Features > Key Features"}, {"id": "ba1f3167e5706e9e", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Quick Start", "content": "Page: Undetected Browser Mode\nSection: Quick Start\n\n```python\nimport asyncio\nfrom crawl4ai import (\n    AsyncWebCrawler, \n    BrowserConfig, \n    CrawlerRunConfig,\n    UndetectedAdapter\n)\nfrom crawl4ai.async_crawler_strategy import…\n```", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import (\n    AsyncWebCrawler, \n    BrowserConfig, \n    CrawlerRunConfig,\n    UndetectedAdapter\n)\nfrom crawl4ai.async_crawler_strategy import…", "filename": ""}], "chunk_position": 17, "heading_path": "Quick Start > Quick Start", "breadcrumbs": "Undetected Browser Mode > Quick Start > Quick Start"}, {"id": "9be345ce1822f4ee", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Combining Both Features", "content": "Page: Undetected Browser Mode\nSection: Combining Both Features\n\nFor maximum evasion, combine stealth mode with undetected browser:", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler, BrowserConfig, UndetectedAdapter\nfrom crawl4ai.async_crawler_strategy import AsyncPlaywrightCrawlerStrategy\n\n# Create browser config with stealth…", "filename": ""}], "chunk_position": 17, "heading_path": "Combining Both Features > Combining Both Features", "breadcrumbs": "Undetected Browser Mode > Combining Both Features > Combining Both Features"}, {"id": "ac41428603416770", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Example 1: Basic Stealth Mode", "content": "Page: Undetected Browser Mode\nSection: Example 1: Basic Stealth Mode\n\n```python\nimport asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig\n\nasync def test_stealth_mode():\n    # Simple stealth mode configuration\n    browser_config = BrowserConfig(…\n```", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig\n\nasync def test_stealth_mode():\n    # Simple stealth mode configuration\n    browser_config = BrowserConfig(…", "filename": ""}], "chunk_position": 17, "heading_path": "Example 1: Basic Stealth Mode > Example 1: Basic Stealth Mode", "breadcrumbs": "Undetected Browser Mode > Example 1: Basic Stealth Mode > Example 1: Basic Stealth Mode"}, {"id": "85064e8517d50176", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Example 2: Undetected Browser Mode", "content": "Page: Undetected Browser Mode\nSection: Example 2: Undetected Browser Mode\n\n```python\nimport asyncio\nfrom crawl4ai import (\n    AsyncWebCrawler,\n    BrowserConfig,\n    CrawlerRunConfig,\n    UndetectedAdapter\n)\nfrom crawl4ai.async_crawler_strategy import…\n```", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import (\n    AsyncWebCrawler,\n    BrowserConfig,\n    CrawlerRunConfig,\n    UndetectedAdapter\n)\nfrom crawl4ai.async_crawler_strategy import…", "filename": ""}], "chunk_position": 17, "heading_path": "Example 2: Undetected Browser Mode > Example 2: Undetected Browser Mode", "breadcrumbs": "Undetected Browser Mode > Example 2: Undetected Browser Mode > Example 2: Undetected Browser Mode"}, {"id": "94f9af24f85b5376", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Browser Adapter Pattern", "content": "Page: Undetected Browser Mode\nSection: Browser Adapter Pattern\n\nThe undetected browser support is implemented using an adapter pattern, allowing seamless switching between different browser implementations:\n\nThe adapter handles:\n- JavaScript execution\n- Console…", "code_blocks": [{"language": "python", "code": "# Regular browser adapter (default)\nfrom crawl4ai import PlaywrightAdapter\nregular_adapter = PlaywrightAdapter()\n\n# Undetected browser adapter\nfrom crawl4ai import…", "filename": ""}], "chunk_position": 17, "heading_path": "Browser Adapter Pattern > Browser Adapter Pattern", "breadcrumbs": "Undetected Browser Mode > Browser Adapter Pattern > Browser Adapter Pattern"}, {"id": "2258d727b06699de", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Best Practices", "content": "Page: Undetected Browser Mode\nSection: Best Practices\n\n- **Avoid Headless Mode**: Detection is easier in headless mode\n- **Use Reasonable Delays**: Don't rush through pages\n- **Rotate User Agents**: You can customize user agents\n- **Handle Failures…", "code_blocks": [{"language": "python", "code": "browser_config = BrowserConfig(headless=False)", "filename": ""}, {"language": "python", "code": "crawler_config = CrawlerRunConfig(\n    wait_time=3.0,  # Wait 3 seconds after page load\n    delay_before_return_html=2.0  # Additional delay\n)", "filename": ""}, {"language": "python", "code": "browser_config = BrowserConfig(\n    headers={\"User-Agent\": \"your-user-agent\"}\n)", "filename": ""}, {"language": "python", "code": "if not result.success:\n    print(f\"Crawl failed: {result.error_message}\")", "filename": ""}], "chunk_position": 17, "heading_path": "Best Practices > Best Practices", "breadcrumbs": "Undetected Browser Mode > Best Practices > Best Practices"}, {"id": "8395871bc32a40de", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Progressive Detection Handling", "content": "Page: Undetected Browser Mode\nSection: Progressive Detection Handling\n\n```python\nasync def crawl_with_progressive_evasion(url):\n    # Step 1: Try regular browser with stealth\n    browser_config = BrowserConfig(\n        enable_stealth=True,\n        headless=False\n    )\n\n    async…\n```", "code_blocks": [{"language": "python", "code": "async def crawl_with_progressive_evasion(url):\n    # Step 1: Try regular browser with stealth\n    browser_config = BrowserConfig(\n        enable_stealth=True,\n        headless=False\n    )\n\n    async…", "filename": ""}], "chunk_position": 17, "heading_path": "Progressive Detection Handling > Progressive Detection Handling", "breadcrumbs": "Undetected Browser Mode > Progressive Detection Handling > Progressive Detection Handling"}, {"id": "f3b148baf7ddd052", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Installation", "content": "Page: Undetected Browser Mode\nSection: Installation\n\nThe undetected browser dependencies are automatically installed when you run:\n\nThis command installs all necessary browser dependencies for both regular and undetected modes.", "code_blocks": [{"language": "bash", "code": "crawl4ai-setup", "filename": ""}], "chunk_position": 17, "heading_path": "Installation > Installation", "breadcrumbs": "Undetected Browser Mode > Installation > Installation"}, {"id": "885e800b40ae4bf9", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Limitations", "content": "Page: Undetected Browser Mode\nSection: Limitations\n\n- **Performance**: Slightly slower than regular mode due to additional patches\n- **Headless Detection**: Some sites can still detect headless mode\n- **Resource Usage**: May use more resources than…", "code_blocks": [], "chunk_position": 17, "heading_path": "Limitations > Limitations", "breadcrumbs": "Undetected Browser Mode > Limitations > Limitations"}, {"id": "ec8d7ae6d62f9e65", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Future Plans", "content": "Page: Undetected Browser Mode\nSection: Future Plans\n\n**Note**: In future versions of Crawl4AI, we may enable stealth mode and undetected browser by default to provide better out-of-the-box success rates. For now, users should explicitly enable these…", "code_blocks": [], "chunk_position": 17, "heading_path": "Future Plans > Future Plans", "breadcrumbs": "Undetected Browser Mode > Future Plans > Future Plans"}, {"id": "8a9e5afe3c4047f4", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "Conclusion", "content": "Page: Undetected Browser Mode\nSection: Conclusion\n\nCrawl4AI provides flexible anti-bot solutions:\n\n- **Start Simple**: Use regular browser + stealth mode for most sites\n- **Escalate if Needed**: Switch to undetected browser for sophisticated…", "code_blocks": [], "chunk_position": 17, "heading_path": "Conclusion > Conclusion", "breadcrumbs": "Undetected Browser Mode > Conclusion > Conclusion"}, {"id": "63cd3e9ea1b5075c", "url": "https://docs.crawl4ai.com/advanced/undetected-browser/", "page_title": "Undetected Browser Mode", "page_type": "guide", "page_summary": "This guide covers Crawl4AI's anti-bot features: Stealth Mode and Undetected Browser Mode, including how to use them, when to use each, and best practices for evading bot detection.", "heading": "See Also", "content": "Page: Undetected Browser Mode\nSection: See Also\n\n- [Advanced Features](../advanced-features/) - Overview of all advanced features\n- [Proxy & Security](../proxy-security/) - Using proxies with anti-bot features\n- [Session…", "code_blocks": [], "chunk_position": 17, "heading_path": "See Also > See Also", "breadcrumbs": "Undetected Browser Mode > See Also > See Also"}, {"id": "9db93c7e3f13f801", "url": "https://docs.crawl4ai.com/advanced/virtual-scroll/", "page_title": "Virtual Scroll - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains Crawl4AI's Virtual Scroll feature for handling virtual scrolling websites, covering configuration, usage examples, comparison with scan_full_page, and performance tips.", "heading": "Understanding Virtual Scroll", "content": "Page: Virtual Scroll - Crawl4AI Documentation (v0.9.x)\nSection: Understanding Virtual Scroll\n\nModern websites increasingly use virtual scrolling (also called windowed rendering or viewport rendering) to handle large datasets efficiently. This technique only renders visible items in the DOM,…", "code_blocks": [{"language": "text", "code": "Traditional Scroll:          Virtual Scroll:\n┌─────────────┐             ┌─────────────┐\n│ Item 1      │             │ Item 11     │  <- Items 1-10 removed\n│ Item 2      │             │ Item 12     │…", "filename": ""}], "chunk_position": 18, "heading_path": "Understanding Virtual Scroll > Understanding Virtual Scroll", "breadcrumbs": "Virtual Scroll - Crawl4AI Documentation (v0.9.x) > Understanding Virtual Scroll > Understanding Virtual Scroll"}, {"id": "a1074de87e8d2b89", "url": "https://docs.crawl4ai.com/advanced/virtual-scroll/", "page_title": "Virtual Scroll - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains Crawl4AI's Virtual Scroll feature for handling virtual scrolling websites, covering configuration, usage examples, comparison with scan_full_page, and performance tips.", "heading": "Basic Usage", "content": "Page: Virtual Scroll - Crawl4AI Documentation (v0.9.x)\nSection: Basic Usage\n\n```python\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, VirtualScrollConfig\n\n# Configure virtual scroll\nvirtual_config = VirtualScrollConfig(\n    container_selector=\"#feed\",      # CSS selector for…\n```", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler, CrawlerRunConfig, VirtualScrollConfig\n\n# Configure virtual scroll\nvirtual_config = VirtualScrollConfig(\n    container_selector=\"#feed\",      # CSS selector for…", "filename": ""}], "chunk_position": 18, "heading_path": "Basic Usage > Basic Usage", "breadcrumbs": "Virtual Scroll - Crawl4AI Documentation (v0.9.x) > Basic Usage > Basic Usage"}, {"id": "36036207a5cba81c", "url": "https://docs.crawl4ai.com/advanced/virtual-scroll/", "page_title": "Virtual Scroll - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains Crawl4AI's Virtual Scroll feature for handling virtual scrolling websites, covering configuration, usage examples, comparison with scan_full_page, and performance tips.", "heading": "Configuration Parameters", "content": "Page: Virtual Scroll - Crawl4AI Documentation (v0.9.x)\nSection: Configuration Parameters\n\n###### VirtualScrollConfig\n\n| Parameter | Type | Default | Description |\n| --- | --- | --- | --- |\n| `container_selector` | `str` | Required | CSS selector for the scrollable container |\n|…", "code_blocks": [], "chunk_position": 18, "heading_path": "Configuration Parameters > Configuration Parameters", "breadcrumbs": "Virtual Scroll - Crawl4AI Documentation (v0.9.x) > Configuration Parameters > Configuration Parameters"}, {"id": "63128d659bccc862", "url": "https://docs.crawl4ai.com/advanced/virtual-scroll/", "page_title": "Virtual Scroll - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains Crawl4AI's Virtual Scroll feature for handling virtual scrolling websites, covering configuration, usage examples, comparison with scan_full_page, and performance tips.", "heading": "Real-World Examples", "content": "Page: Virtual Scroll - Crawl4AI Documentation (v0.9.x)\nSection: Real-World Examples\n\n###### Twitter-like Timeline\n\nTwitter replaces tweets as you scroll.\n\n###### Instagram Grid\n\nInstagram uses virtualized grid for performance.\n\n###### Mixed Content (News Feed)\n\nSome sites mix static and…", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler, CrawlerRunConfig, VirtualScrollConfig, BrowserConfig\n\nasync def crawl_twitter_timeline():\n    # Twitter replaces tweets as you scroll\n    virtual_config =…", "filename": ""}, {"language": "python", "code": "async def crawl_instagram_grid():\n    # Instagram uses virtualized grid for performance\n    virtual_config = VirtualScrollConfig(\n        container_selector=\"article\",  # Main feed container…", "filename": ""}, {"language": "python", "code": "async def crawl_mixed_feed():\n    # Featured articles stay, regular articles virtualize\n    virtual_config = VirtualScrollConfig(\n        container_selector=\".main-feed\",\n        scroll_count=25,…", "filename": ""}], "chunk_position": 18, "heading_path": "Real-World Examples > Real-World Examples", "breadcrumbs": "Virtual Scroll - Crawl4AI Documentation (v0.9.x) > Real-World Examples > Real-World Examples"}, {"id": "36ea652549c2aa1f", "url": "https://docs.crawl4ai.com/advanced/virtual-scroll/", "page_title": "Virtual Scroll - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains Crawl4AI's Virtual Scroll feature for handling virtual scrolling websites, covering configuration, usage examples, comparison with scan_full_page, and performance tips.", "heading": "Virtual Scroll vs scan_full_page", "content": "Page: Virtual Scroll - Crawl4AI Documentation (v0.9.x)\nSection: Virtual Scroll vs scan_full_page\n\nBoth features handle dynamic content, but serve different purposes:\n\n| Feature | Virtual Scroll | scan_full_page |\n| --- | --- | --- |\n| **Purpose** | Capture content that's replaced during scroll |…", "code_blocks": [], "chunk_position": 18, "heading_path": "Virtual Scroll vs scan_full_page > Virtual Scroll vs scan_full_page", "breadcrumbs": "Virtual Scroll - Crawl4AI Documentation (v0.9.x) > Virtual Scroll vs scan_full_page > Virtual Scroll vs scan_full_page"}, {"id": "79df76d9237a4d08", "url": "https://docs.crawl4ai.com/advanced/virtual-scroll/", "page_title": "Virtual Scroll - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains Crawl4AI's Virtual Scroll feature for handling virtual scrolling websites, covering configuration, usage examples, comparison with scan_full_page, and performance tips.", "heading": "Combining with Extraction", "content": "Page: Virtual Scroll - Crawl4AI Documentation (v0.9.x)\nSection: Combining with Extraction\n\nVirtual Scroll works seamlessly with extraction strategies.", "code_blocks": [{"language": "python", "code": "from crawl4ai import LLMExtractionStrategy, LLMConfig\n\n# Define extraction schema\nschema = {\n    \"type\": \"array\",\n    \"items\": {\n        \"type\": \"object\", \n        \"properties\": {…", "filename": ""}], "chunk_position": 18, "heading_path": "Combining with Extraction > Combining with Extraction", "breadcrumbs": "Virtual Scroll - Crawl4AI Documentation (v0.9.x) > Combining with Extraction > Combining with Extraction"}, {"id": "c1eab0cff2af0c8b", "url": "https://docs.crawl4ai.com/advanced/virtual-scroll/", "page_title": "Virtual Scroll - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains Crawl4AI's Virtual Scroll feature for handling virtual scrolling websites, covering configuration, usage examples, comparison with scan_full_page, and performance tips.", "heading": "Performance Tips", "content": "Page: Virtual Scroll - Crawl4AI Documentation (v0.9.x)\nSection: Performance Tips\n\n- **Container Selection**: Be specific with selectors. Using the correct container improves performance.\n- **Scroll Count**: Start conservative and increase as needed.\n- **Wait Times**: Adjust based…", "code_blocks": [{"language": "python", "code": "# Start with fewer scrolls\nvirtual_config = VirtualScrollConfig(\n    container_selector=\"#feed\",\n    scroll_count=10  # Test with 10, increase if needed\n)", "filename": ""}, {"language": "python", "code": "# Fast sites\nwait_after_scroll=0.2\n\n# Slower sites or heavy content\nwait_after_scroll=1.5", "filename": ""}, {"language": "python", "code": "browser_config = BrowserConfig(headless=False)\nasync with AsyncWebCrawler(config=browser_config) as crawler:\n    # Watch the scrolling happen", "filename": ""}], "chunk_position": 18, "heading_path": "Performance Tips > Performance Tips", "breadcrumbs": "Virtual Scroll - Crawl4AI Documentation (v0.9.x) > Performance Tips > Performance Tips"}, {"id": "147662e2ed1be2cf", "url": "https://docs.crawl4ai.com/advanced/virtual-scroll/", "page_title": "Virtual Scroll - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains Crawl4AI's Virtual Scroll feature for handling virtual scrolling websites, covering configuration, usage examples, comparison with scan_full_page, and performance tips.", "heading": "How It Works Internally", "content": "Page: Virtual Scroll - Crawl4AI Documentation (v0.9.x)\nSection: How It Works Internally\n\n- **Detection Phase**: Scrolls and compares HTML to detect behavior\n- **Capture Phase**: For replaced content, stores HTML chunks at each position\n- **Merge Phase**: Combines all chunks, removing…", "code_blocks": [], "chunk_position": 18, "heading_path": "How It Works Internally > How It Works Internally", "breadcrumbs": "Virtual Scroll - Crawl4AI Documentation (v0.9.x) > How It Works Internally > How It Works Internally"}, {"id": "de25cd3a59fe901a", "url": "https://docs.crawl4ai.com/advanced/virtual-scroll/", "page_title": "Virtual Scroll - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains Crawl4AI's Virtual Scroll feature for handling virtual scrolling websites, covering configuration, usage examples, comparison with scan_full_page, and performance tips.", "heading": "Error Handling", "content": "Page: Virtual Scroll - Crawl4AI Documentation (v0.9.x)\nSection: Error Handling\n\nVirtual Scroll handles errors gracefully. If the container isn't found, crawling continues normally without virtual scroll.", "code_blocks": [{"language": "python", "code": "# If container not found or scrolling fails\nresult = await crawler.arun(url=\"...\", config=config)\n\nif result.success:\n    # Virtual scroll worked or wasn't needed\n    print(f\"Captured…", "filename": ""}], "chunk_position": 18, "heading_path": "Error Handling > Error Handling", "breadcrumbs": "Virtual Scroll - Crawl4AI Documentation (v0.9.x) > Error Handling > Error Handling"}, {"id": "d7937ac32655bed0", "url": "https://docs.crawl4ai.com/advanced/virtual-scroll/", "page_title": "Virtual Scroll - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains Crawl4AI's Virtual Scroll feature for handling virtual scrolling websites, covering configuration, usage examples, comparison with scan_full_page, and performance tips.", "heading": "Complete Example", "content": "Page: Virtual Scroll - Crawl4AI Documentation (v0.9.x)\nSection: Complete Example\n\nSee our comprehensive example that demonstrates:\n- Twitter-like feeds\n- Instagram grids\n- Traditional infinite scroll\n- Mixed content scenarios\n- Performance comparisons\n\nThe example includes a local…", "code_blocks": [{"language": "bash", "code": "# Run the examples\ncd docs/examples\npython virtual_scroll_example.py", "filename": ""}], "chunk_position": 18, "heading_path": "Complete Example > Complete Example", "breadcrumbs": "Virtual Scroll - Crawl4AI Documentation (v0.9.x) > Complete Example > Complete Example"}, {"id": "6f7532e6ac9a6f8b", "url": "https://docs.crawl4ai.com/api/adaptive-crawler/", "page_title": "AdaptiveCrawler", "page_type": "api", "page_summary": "The AdaptiveCrawler class implements intelligent web crawling that automatically determines when sufficient information has been gathered to answer a query. It uses a three-layer scoring system to…", "heading": "Constructor", "content": "Page: AdaptiveCrawler\nSection: Constructor\n\n###### Parameters\n\n- **crawler**  (`AsyncWebCrawler`): The underlying web crawler instance to use for fetching pages\n- **config**  (`Optional[AdaptiveConfig]`): Configuration settings for adaptive…", "code_blocks": [{"language": "python", "code": "AdaptiveCrawler(\n    crawler: AsyncWebCrawler,\n    config: Optional[AdaptiveConfig] = None\n)", "filename": ""}], "chunk_position": 19, "heading_path": "Constructor > Constructor", "breadcrumbs": "AdaptiveCrawler > Constructor > Constructor"}, {"id": "64cde89ef5e9e1cc", "url": "https://docs.crawl4ai.com/api/adaptive-crawler/", "page_title": "AdaptiveCrawler", "page_type": "api", "page_summary": "The AdaptiveCrawler class implements intelligent web crawling that automatically determines when sufficient information has been gathered to answer a query. It uses a three-layer scoring system to…", "heading": "Primary Method", "content": "Page: AdaptiveCrawler\nSection: Primary Method\n\n###### digest()\n\nThe main method that performs adaptive crawling starting from a URL with a specific query.\n\n###### Parameters\n\n- **start_url**  (`str`): The starting URL for crawling\n- **query**…", "code_blocks": [{"language": "python", "code": "async def digest(\n    start_url: str,\n    query: str,\n    resume_from: Optional[Union[str, Path]] = None\n) -> CrawlState", "filename": ""}, {"language": "python", "code": "async with AsyncWebCrawler() as crawler:\n    adaptive = AdaptiveCrawler(crawler)\n    state = await adaptive.digest(\n        start_url=\"https://docs.python.org\",\n        query=\"async context…", "filename": ""}], "chunk_position": 19, "heading_path": "Primary Method > Primary Method", "breadcrumbs": "AdaptiveCrawler > Primary Method > Primary Method"}, {"id": "bea9709d5c4574b8", "url": "https://docs.crawl4ai.com/api/adaptive-crawler/", "page_title": "AdaptiveCrawler", "page_type": "api", "page_summary": "The AdaptiveCrawler class implements intelligent web crawling that automatically determines when sufficient information has been gathered to answer a query. It uses a three-layer scoring system to…", "heading": "Properties", "content": "Page: AdaptiveCrawler\nSection: Properties\n\n###### confidence\n\nCurrent confidence score (0-1) indicating information sufficiency.\n\n###### coverage_stats\n\nDictionary containing detailed coverage statistics.\n\nReturns:\n-  **coverage** : Query term…", "code_blocks": [{"language": "python", "code": "@property\ndef confidence(self) -> float", "filename": ""}, {"language": "python", "code": "@property  \ndef coverage_stats(self) -> Dict[str, float]", "filename": ""}, {"language": "python", "code": "@property\ndef is_sufficient(self) -> bool", "filename": ""}, {"language": "python", "code": "@property\ndef state(self) -> CrawlState", "filename": ""}], "chunk_position": 19, "heading_path": "Properties > Properties", "breadcrumbs": "AdaptiveCrawler > Properties > Properties"}, {"id": "2b3e42f67b20a081", "url": "https://docs.crawl4ai.com/api/adaptive-crawler/", "page_title": "AdaptiveCrawler", "page_type": "api", "page_summary": "The AdaptiveCrawler class implements intelligent web crawling that automatically determines when sufficient information has been gathered to answer a query. It uses a three-layer scoring system to…", "heading": "Methods", "content": "Page: AdaptiveCrawler\nSection: Methods\n\n###### get_relevant_content()\n\nRetrieve the most relevant content from the knowledge base.\n\n###### Parameters\n\n- **top_k**  (`int`): Number of top relevant documents to return (default: 5)\n\n####…", "code_blocks": [{"language": "python", "code": "def get_relevant_content(\n    self,\n    top_k: int = 5\n) -> List[Dict[str, Any]]", "filename": ""}, {"language": "python", "code": "def print_stats(\n    self,\n    detailed: bool = False\n) -> None", "filename": ""}, {"language": "python", "code": "def export_knowledge_base(\n    self,\n    path: Union[str, Path]\n) -> None", "filename": ""}, {"language": "python", "code": "adaptive.export_knowledge_base(\"my_knowledge.jsonl\")", "filename": ""}, {"language": "python", "code": "async def import_knowledge_base(\n    self,\n    path: Union[str, Path]\n) -> None", "filename": ""}], "chunk_position": 19, "heading_path": "Methods > Methods", "breadcrumbs": "AdaptiveCrawler > Methods > Methods"}, {"id": "ac0a61715851133a", "url": "https://docs.crawl4ai.com/api/adaptive-crawler/", "page_title": "AdaptiveCrawler", "page_type": "api", "page_summary": "The AdaptiveCrawler class implements intelligent web crawling that automatically determines when sufficient information has been gathered to answer a query. It uses a three-layer scoring system to…", "heading": "Configuration", "content": "Page: AdaptiveCrawler\nSection: Configuration\n\nThe `AdaptiveConfig` class controls the behavior of adaptive crawling:\n\n###### Example with Custom Config", "code_blocks": [{"language": "python", "code": "@dataclass\nclass AdaptiveConfig:\n    confidence_threshold: float = 0.8      # Stop when confidence reaches this\n    max_pages: int = 50                    # Maximum pages to crawl\n    top_k_links:…", "filename": ""}, {"language": "python", "code": "config = AdaptiveConfig(\n    confidence_threshold=0.7,\n    max_pages=20,\n    top_k_links=3\n)\n\nadaptive = AdaptiveCrawler(crawler, config=config)", "filename": ""}], "chunk_position": 19, "heading_path": "Configuration > Configuration", "breadcrumbs": "AdaptiveCrawler > Configuration > Configuration"}, {"id": "f8be191da94b7cb6", "url": "https://docs.crawl4ai.com/api/adaptive-crawler/", "page_title": "AdaptiveCrawler", "page_type": "api", "page_summary": "The AdaptiveCrawler class implements intelligent web crawling that automatically determines when sufficient information has been gathered to answer a query. It uses a three-layer scoring system to…", "heading": "Complete Example", "content": "Page: AdaptiveCrawler\nSection: Complete Example\n\n```python\nimport asyncio\nfrom crawl4ai import AsyncWebCrawler, AdaptiveCrawler, AdaptiveConfig\n\nasync def main():\n    # Configure adaptive crawling\n    config = AdaptiveConfig(…\n```", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, AdaptiveCrawler, AdaptiveConfig\n\nasync def main():\n    # Configure adaptive crawling\n    config = AdaptiveConfig(…", "filename": ""}], "chunk_position": 19, "heading_path": "Complete Example > Complete Example", "breadcrumbs": "AdaptiveCrawler > Complete Example > Complete Example"}, {"id": "cd5428d97df7f253", "url": "https://docs.crawl4ai.com/api/adaptive-crawler/", "page_title": "AdaptiveCrawler", "page_type": "api", "page_summary": "The AdaptiveCrawler class implements intelligent web crawling that automatically determines when sufficient information has been gathered to answer a query. It uses a three-layer scoring system to…", "heading": "See Also", "content": "Page: AdaptiveCrawler\nSection: See Also\n\n- [digest() Method Reference](../digest/)\n- [Adaptive Crawling Guide](../../core/adaptive-crawling/)\n- [Advanced Adaptive Strategies](../../advanced/adaptive-strategies/)", "code_blocks": [], "chunk_position": 19, "heading_path": "See Also > See Also", "breadcrumbs": "AdaptiveCrawler > See Also > See Also"}, {"id": "fa396d595ed648e7", "url": "https://docs.crawl4ai.com/api/arun/", "page_title": "`arun()` Parameter Guide (New Approach)", "page_type": "guide", "page_summary": "A guide to the parameters of the `arun()` method in Crawl4AI, now organized under `CrawlerRunConfig`, covering caching, content processing, navigation, session management, media options, extraction,…", "heading": "Introduction", "content": "Page: `arun()` Parameter Guide (New Approach)\nSection: Introduction\n\nIn Crawl4AI's **latest** configuration model, nearly all parameters that once went directly to `arun()` are now part of **`CrawlerRunConfig`** . When calling `arun()`, you provide:\n\nBelow is an…", "code_blocks": [{"language": "python", "code": "await crawler.arun(\n    url=\"https://example.com\",  \n    config=my_run_config\n)", "filename": ""}], "chunk_position": 20, "heading_path": "Introduction > Introduction", "breadcrumbs": "`arun()` Parameter Guide (New Approach) > Introduction > Introduction"}, {"id": "86199df6d29de8c6", "url": "https://docs.crawl4ai.com/api/arun/", "page_title": "`arun()` Parameter Guide (New Approach)", "page_type": "guide", "page_summary": "A guide to the parameters of the `arun()` method in Crawl4AI, now organized under `CrawlerRunConfig`, covering caching, content processing, navigation, session management, media options, extraction,…", "heading": "1. Core Usage", "content": "Page: `arun()` Parameter Guide (New Approach)\nSection: 1. Core Usage\n\n**Key Fields**:\n- `verbose=True` logs each crawl step. \n- `cache_mode` decides how to read/write the local crawl cache.", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\n\nasync def main():\n    run_config = CrawlerRunConfig(\n        verbose=True,            # Detailed logging…", "filename": ""}], "chunk_position": 20, "heading_path": "1. Core Usage > 1. Core Usage", "breadcrumbs": "`arun()` Parameter Guide (New Approach) > 1. Core Usage > 1. Core Usage"}, {"id": "680cdd005a5f7dd3", "url": "https://docs.crawl4ai.com/api/arun/", "page_title": "`arun()` Parameter Guide (New Approach)", "page_type": "guide", "page_summary": "A guide to the parameters of the `arun()` method in Crawl4AI, now organized under `CrawlerRunConfig`, covering caching, content processing, navigation, session management, media options, extraction,…", "heading": "2. Cache Control", "content": "Page: `arun()` Parameter Guide (New Approach)\nSection: 2. Cache Control\n\n**`cache_mode`** (default: `CacheMode.ENABLED`)\n\nUse a built-in enum from `CacheMode`:\n\n- `ENABLED`: Normal caching—reads if available, writes if missing.\n- `DISABLED`: No caching—always refetch…", "code_blocks": [{"language": "python", "code": "run_config = CrawlerRunConfig(\n    cache_mode=CacheMode.BYPASS\n)", "filename": ""}], "chunk_position": 20, "heading_path": "2. Cache Control > 2. Cache Control", "breadcrumbs": "`arun()` Parameter Guide (New Approach) > 2. Cache Control > 2. Cache Control"}, {"id": "6e52470d38f9159e", "url": "https://docs.crawl4ai.com/api/arun/", "page_title": "`arun()` Parameter Guide (New Approach)", "page_type": "guide", "page_summary": "A guide to the parameters of the `arun()` method in Crawl4AI, now organized under `CrawlerRunConfig`, covering caching, content processing, navigation, session management, media options, extraction,…", "heading": "3.1 Text Processing", "content": "Page: `arun()` Parameter Guide (New Approach)\nSection: 3.1 Text Processing\n\n```python\nrun_config = CrawlerRunConfig(\n    word_count_threshold=10,   # Ignore text blocks <10 words\n    only_text=False,           # If True, tries to remove non-text elements\n    keep_data_attributes=False…\n```", "code_blocks": [{"language": "python", "code": "run_config = CrawlerRunConfig(\n    word_count_threshold=10,   # Ignore text blocks <10 words\n    only_text=False,           # If True, tries to remove non-text elements\n    keep_data_attributes=False…", "filename": ""}], "chunk_position": 20, "heading_path": "3.1 Text Processing > 3.1 Text Processing", "breadcrumbs": "`arun()` Parameter Guide (New Approach) > 3.1 Text Processing > 3.1 Text Processing"}, {"id": "9f2c4f8c8e83647b", "url": "https://docs.crawl4ai.com/api/arun/", "page_title": "`arun()` Parameter Guide (New Approach)", "page_type": "guide", "page_summary": "A guide to the parameters of the `arun()` method in Crawl4AI, now organized under `CrawlerRunConfig`, covering caching, content processing, navigation, session management, media options, extraction,…", "heading": "3.2 Content Selection", "content": "Page: `arun()` Parameter Guide (New Approach)\nSection: 3.2 Content Selection\n\n```python\nrun_config = CrawlerRunConfig(\n    css_selector=\".main-content\",  # Focus on .main-content region only\n    excluded_tags=[\"form\", \"nav\"], # Remove entire tag blocks\n    remove_forms=True,…\n```", "code_blocks": [{"language": "python", "code": "run_config = CrawlerRunConfig(\n    css_selector=\".main-content\",  # Focus on .main-content region only\n    excluded_tags=[\"form\", \"nav\"], # Remove entire tag blocks\n    remove_forms=True,…", "filename": ""}], "chunk_position": 20, "heading_path": "3.2 Content Selection > 3.2 Content Selection", "breadcrumbs": "`arun()` Parameter Guide (New Approach) > 3.2 Content Selection > 3.2 Content Selection"}, {"id": "3e0deccb3647208c", "url": "https://docs.crawl4ai.com/api/arun/", "page_title": "`arun()` Parameter Guide (New Approach)", "page_type": "guide", "page_summary": "A guide to the parameters of the `arun()` method in Crawl4AI, now organized under `CrawlerRunConfig`, covering caching, content processing, navigation, session management, media options, extraction,…", "heading": "3.3 Link Handling", "content": "Page: `arun()` Parameter Guide (New Approach)\nSection: 3.3 Link Handling\n\n```python\nrun_config = CrawlerRunConfig(\n    exclude_external_links=True,         # Remove external links from final content\n    exclude_social_media_links=True,     # Remove links to known social sites…\n```", "code_blocks": [{"language": "python", "code": "run_config = CrawlerRunConfig(\n    exclude_external_links=True,         # Remove external links from final content\n    exclude_social_media_links=True,     # Remove links to known social sites…", "filename": ""}], "chunk_position": 20, "heading_path": "3.3 Link Handling > 3.3 Link Handling", "breadcrumbs": "`arun()` Parameter Guide (New Approach) > 3.3 Link Handling > 3.3 Link Handling"}, {"id": "7ae03aee3d778b99", "url": "https://docs.crawl4ai.com/api/arun/", "page_title": "`arun()` Parameter Guide (New Approach)", "page_type": "guide", "page_summary": "A guide to the parameters of the `arun()` method in Crawl4AI, now organized under `CrawlerRunConfig`, covering caching, content processing, navigation, session management, media options, extraction,…", "heading": "3.4 Media Filtering", "content": "Page: `arun()` Parameter Guide (New Approach)\nSection: 3.4 Media Filtering\n\n```python\nrun_config = CrawlerRunConfig(\n    exclude_external_images=True  # Strip images from other domains\n)\n```", "code_blocks": [{"language": "python", "code": "run_config = CrawlerRunConfig(\n    exclude_external_images=True  # Strip images from other domains\n)", "filename": ""}], "chunk_position": 20, "heading_path": "3.4 Media Filtering > 3.4 Media Filtering", "breadcrumbs": "`arun()` Parameter Guide (New Approach) > 3.4 Media Filtering > 3.4 Media Filtering"}, {"id": "5dddbbfa14996c38", "url": "https://docs.crawl4ai.com/api/arun/", "page_title": "`arun()` Parameter Guide (New Approach)", "page_type": "guide", "page_summary": "A guide to the parameters of the `arun()` method in Crawl4AI, now organized under `CrawlerRunConfig`, covering caching, content processing, navigation, session management, media options, extraction,…", "heading": "4.1 Basic Browser Flow", "content": "Page: `arun()` Parameter Guide (New Approach)\nSection: 4.1 Basic Browser Flow\n\n**Key Fields**:\n- `wait_for`:\n  - `\"css:selector\"` or\n  - `\"js:() => boolean\"`\n  e.g. `js:() => document.querySelectorAll('.item').length > 10`.\n- `mean_delay` & `max_range`: define random delays for…", "code_blocks": [{"language": "python", "code": "run_config = CrawlerRunConfig(\n    wait_for=\"css:.dynamic-content\", # Wait for .dynamic-content\n    delay_before_return_html=2.0,    # Wait 2s before capturing final HTML\n    page_timeout=60000,…", "filename": ""}], "chunk_position": 20, "heading_path": "4.1 Basic Browser Flow > 4.1 Basic Browser Flow", "breadcrumbs": "`arun()` Parameter Guide (New Approach) > 4.1 Basic Browser Flow > 4.1 Basic Browser Flow"}, {"id": "ce2aab4732300506", "url": "https://docs.crawl4ai.com/api/arun/", "page_title": "`arun()` Parameter Guide (New Approach)", "page_type": "guide", "page_summary": "A guide to the parameters of the `arun()` method in Crawl4AI, now organized under `CrawlerRunConfig`, covering caching, content processing, navigation, session management, media options, extraction,…", "heading": "4.2 JavaScript Execution", "content": "Page: `arun()` Parameter Guide (New Approach)\nSection: 4.2 JavaScript Execution\n\n- `js_code` can be a single string or a list of strings.\n- `js_only=True` means “I’m continuing in the same session with new JS steps, no new full navigation.”", "code_blocks": [{"language": "python", "code": "run_config = CrawlerRunConfig(\n    js_code=[\n        \"window.scrollTo(0, document.body.scrollHeight);\",\n        \"document.querySelector('.load-more')?.click();\"\n    ],\n    js_only=False\n)", "filename": ""}], "chunk_position": 20, "heading_path": "4.2 JavaScript Execution > 4.2 JavaScript Execution", "breadcrumbs": "`arun()` Parameter Guide (New Approach) > 4.2 JavaScript Execution > 4.2 JavaScript Execution"}, {"id": "faef4bcf7debac42", "url": "https://docs.crawl4ai.com/api/arun/", "page_title": "`arun()` Parameter Guide (New Approach)", "page_type": "guide", "page_summary": "A guide to the parameters of the `arun()` method in Crawl4AI, now organized under `CrawlerRunConfig`, covering caching, content processing, navigation, session management, media options, extraction,…", "heading": "4.3 Anti-Bot", "content": "Page: `arun()` Parameter Guide (New Approach)\nSection: 4.3 Anti-Bot\n\n- `magic=True` tries multiple stealth features. \n- `simulate_user=True` mimics mouse movements or random delays. \n- `override_navigator=True` fakes some navigator properties (like user agent checks).", "code_blocks": [{"language": "python", "code": "run_config = CrawlerRunConfig(\n    magic=True,\n    simulate_user=True,\n    override_navigator=True\n)", "filename": ""}], "chunk_position": 20, "heading_path": "4.3 Anti-Bot > 4.3 Anti-Bot", "breadcrumbs": "`arun()` Parameter Guide (New Approach) > 4.3 Anti-Bot > 4.3 Anti-Bot"}, {"id": "1ea236ef6ce45859", "url": "https://docs.crawl4ai.com/api/arun/", "page_title": "`arun()` Parameter Guide (New Approach)", "page_type": "guide", "page_summary": "A guide to the parameters of the `arun()` method in Crawl4AI, now organized under `CrawlerRunConfig`, covering caching, content processing, navigation, session management, media options, extraction,…", "heading": "5. Session Management", "content": "Page: `arun()` Parameter Guide (New Approach)\nSection: 5. Session Management\n\nIf re-used in subsequent `arun()` calls, the same tab/page context is continued (helpful for multi-step tasks or stateful browsing).", "code_blocks": [{"language": "python", "code": "run_config = CrawlerRunConfig(\n    session_id=\"my_session123\"\n)", "filename": ""}], "chunk_position": 20, "heading_path": "5. Session Management > 5. Session Management", "breadcrumbs": "`arun()` Parameter Guide (New Approach) > 5. Session Management > 5. Session Management"}, {"id": "b13c36f4a7411de4", "url": "https://docs.crawl4ai.com/api/arun/", "page_title": "`arun()` Parameter Guide (New Approach)", "page_type": "guide", "page_summary": "A guide to the parameters of the `arun()` method in Crawl4AI, now organized under `CrawlerRunConfig`, covering caching, content processing, navigation, session management, media options, extraction,…", "heading": "6. Screenshot, PDF & Media Options", "content": "Page: `arun()` Parameter Guide (New Approach)\nSection: 6. Screenshot, PDF & Media Options\n\n**Where they appear**:\n- `result.screenshot` → Base64 screenshot string.\n- `result.pdf` → Byte array with PDF data.", "code_blocks": [{"language": "python", "code": "run_config = CrawlerRunConfig(\n    screenshot=True,             # Grab a screenshot as base64\n    screenshot_wait_for=1.0,     # Wait 1s before capturing\n    pdf=True,                    # Also…", "filename": ""}], "chunk_position": 20, "heading_path": "6. Screenshot, PDF & Media Options > 6. Screenshot, PDF & Media Options", "breadcrumbs": "`arun()` Parameter Guide (New Approach) > 6. Screenshot, PDF & Media Options > 6. Screenshot, PDF & Media Options"}, {"id": "334148989f3de671", "url": "https://docs.crawl4ai.com/api/arun/", "page_title": "`arun()` Parameter Guide (New Approach)", "page_type": "guide", "page_summary": "A guide to the parameters of the `arun()` method in Crawl4AI, now organized under `CrawlerRunConfig`, covering caching, content processing, navigation, session management, media options, extraction,…", "heading": "7. Extraction Strategy", "content": "Page: `arun()` Parameter Guide (New Approach)\nSection: 7. Extraction Strategy\n\nThe extracted data will appear in `result.extracted_content`.", "code_blocks": [{"language": "python", "code": "run_config = CrawlerRunConfig(\n    extraction_strategy=my_css_or_llm_strategy\n)", "filename": ""}], "chunk_position": 20, "heading_path": "7. Extraction Strategy > 7. Extraction Strategy", "breadcrumbs": "`arun()` Parameter Guide (New Approach) > 7. Extraction Strategy > 7. Extraction Strategy"}, {"id": "67824531624dde96", "url": "https://docs.crawl4ai.com/api/arun/", "page_title": "`arun()` Parameter Guide (New Approach)", "page_type": "guide", "page_summary": "A guide to the parameters of the `arun()` method in Crawl4AI, now organized under `CrawlerRunConfig`, covering caching, content processing, navigation, session management, media options, extraction,…", "heading": "8. Comprehensive Example", "content": "Page: `arun()` Parameter Guide (New Approach)\nSection: 8. Comprehensive Example\n\nBelow is a snippet combining many parameters:\n\n**What we covered**:\n1. **Crawling** the main content region, ignoring external links. \n2. Running **JavaScript** to click “.show-more”. \n3. **Waiting**…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\nfrom crawl4ai import JsonCssExtractionStrategy\n\nasync def main():\n    # Example schema\n    schema = {\n        \"name\":…", "filename": ""}], "chunk_position": 20, "heading_path": "8. Comprehensive Example > 8. Comprehensive Example", "breadcrumbs": "`arun()` Parameter Guide (New Approach) > 8. Comprehensive Example > 8. Comprehensive Example"}, {"id": "2bfcb3c30e3bba39", "url": "https://docs.crawl4ai.com/api/arun/", "page_title": "`arun()` Parameter Guide (New Approach)", "page_type": "guide", "page_summary": "A guide to the parameters of the `arun()` method in Crawl4AI, now organized under `CrawlerRunConfig`, covering caching, content processing, navigation, session management, media options, extraction,…", "heading": "9. Best Practices", "content": "Page: `arun()` Parameter Guide (New Approach)\nSection: 9. Best Practices\n\n1. **Use `BrowserConfig` for global browser** settings (headless, user agent). \n2. **Use `CrawlerRunConfig`** to handle the **specific** crawl needs: content filtering, caching, JS, screenshot,…", "code_blocks": [], "chunk_position": 20, "heading_path": "9. Best Practices > 9. Best Practices", "breadcrumbs": "`arun()` Parameter Guide (New Approach) > 9. Best Practices > 9. Best Practices"}, {"id": "eab982906cb81e73", "url": "https://docs.crawl4ai.com/api/arun/", "page_title": "`arun()` Parameter Guide (New Approach)", "page_type": "guide", "page_summary": "A guide to the parameters of the `arun()` method in Crawl4AI, now organized under `CrawlerRunConfig`, covering caching, content processing, navigation, session management, media options, extraction,…", "heading": "10. Conclusion", "content": "Page: `arun()` Parameter Guide (New Approach)\nSection: 10. Conclusion\n\nAll parameters that used to be direct arguments to `arun()` now belong in **`CrawlerRunConfig`** . This approach:\n\n- Makes code **clearer** and **more maintainable**.\n- Minimizes confusion about…", "code_blocks": [], "chunk_position": 20, "heading_path": "10. Conclusion > 10. Conclusion", "breadcrumbs": "`arun()` Parameter Guide (New Approach) > 10. Conclusion > 10. Conclusion"}, {"id": "c780e69220e33492", "url": "https://docs.crawl4ai.com/api/arun_many/", "page_title": "arun_many()", "page_type": "api", "page_summary": "Reference for the arun_many() function in Crawl4AI, which crawls multiple URLs concurrently or in batches, with support for dispatchers, streaming, and per-URL configurations.", "heading": "Function Signature", "content": "Page: arun_many()\nSection: Function Signature\n\n```python\nasync def arun_many(\n    urls: Union[List[str], List[Any]],\n    config: Optional[Union[CrawlerRunConfig, List[CrawlerRunConfig]]] = None,\n    dispatcher: Optional[BaseDispatcher] = None,\n    ...\n) ->…\n```", "code_blocks": [{"language": "python", "code": "async def arun_many(\n    urls: Union[List[str], List[Any]],\n    config: Optional[Union[CrawlerRunConfig, List[CrawlerRunConfig]]] = None,\n    dispatcher: Optional[BaseDispatcher] = None,\n    ...\n) ->…", "filename": ""}], "chunk_position": 21, "heading_path": "Function Signature > Function Signature", "breadcrumbs": "arun_many() > Function Signature > Function Signature"}, {"id": "d6133a685af6630f", "url": "https://docs.crawl4ai.com/api/arun_many/", "page_title": "arun_many()", "page_type": "api", "page_summary": "Reference for the arun_many() function in Crawl4AI, which crawls multiple URLs concurrently or in batches, with support for dispatchers, streaming, and per-URL configurations.", "heading": "Differences from arun()", "content": "Page: arun_many()\nSection: Differences from arun()\n\n1. **Multiple URLs**:\n- Instead of crawling a single URL, you pass a list of them (strings or tasks).\n- The function returns `RunManyReturn` which contains either a **list** of `CrawlResult` or an…", "code_blocks": [], "chunk_position": 21, "heading_path": "Differences from arun() > Differences from arun()", "breadcrumbs": "arun_many() > Differences from arun() > Differences from arun()"}, {"id": "7bab31ffac738286", "url": "https://docs.crawl4ai.com/api/arun_many/", "page_title": "arun_many()", "page_type": "api", "page_summary": "Reference for the arun_many() function in Crawl4AI, which crawls multiple URLs concurrently or in batches, with support for dispatchers, streaming, and per-URL configurations.", "heading": "Basic Example (Batch Mode)", "content": "Page: arun_many()\nSection: Basic Example (Batch Mode)\n\n```python\n# Minimal usage: The default dispatcher will be used\nresults = await crawler.arun_many(\n    urls=[\"https://site1.com\", \"https://site2.com\"],\n    config=CrawlerRunConfig(stream=False)  # Default…\n```", "code_blocks": [{"language": "python", "code": "# Minimal usage: The default dispatcher will be used\nresults = await crawler.arun_many(\n    urls=[\"https://site1.com\", \"https://site2.com\"],\n    config=CrawlerRunConfig(stream=False)  # Default…", "filename": ""}], "chunk_position": 21, "heading_path": "Basic Example (Batch Mode) > Basic Example (Batch Mode)", "breadcrumbs": "arun_many() > Basic Example (Batch Mode) > Basic Example (Batch Mode)"}, {"id": "ef4c755de67d05db", "url": "https://docs.crawl4ai.com/api/arun_many/", "page_title": "arun_many()", "page_type": "api", "page_summary": "Reference for the arun_many() function in Crawl4AI, which crawls multiple URLs concurrently or in batches, with support for dispatchers, streaming, and per-URL configurations.", "heading": "Streaming Example", "content": "Page: arun_many()\nSection: Streaming Example\n\n```python\nconfig = CrawlerRunConfig(\n    stream=True,  # Enable streaming mode\n    cache_mode=CacheMode.BYPASS\n)\n\n# Process results as they complete\nasync for result in await crawler.arun_many(…\n```", "code_blocks": [{"language": "python", "code": "config = CrawlerRunConfig(\n    stream=True,  # Enable streaming mode\n    cache_mode=CacheMode.BYPASS\n)\n\n# Process results as they complete\nasync for result in await crawler.arun_many(…", "filename": ""}], "chunk_position": 21, "heading_path": "Streaming Example > Streaming Example", "breadcrumbs": "arun_many() > Streaming Example > Streaming Example"}, {"id": "d5bd06488ea76867", "url": "https://docs.crawl4ai.com/api/arun_many/", "page_title": "arun_many()", "page_type": "api", "page_summary": "Reference for the arun_many() function in Crawl4AI, which crawls multiple URLs concurrently or in batches, with support for dispatchers, streaming, and per-URL configurations.", "heading": "With a Custom Dispatcher", "content": "Page: arun_many()\nSection: With a Custom Dispatcher\n\n```python\ndispatcher = MemoryAdaptiveDispatcher(\n    memory_threshold_percent=70.0,\n    max_session_permit=10\n)\nresults = await crawler.arun_many(\n    urls=[\"https://site1.com\", \"https://site2.com\",…\n```", "code_blocks": [{"language": "python", "code": "dispatcher = MemoryAdaptiveDispatcher(\n    memory_threshold_percent=70.0,\n    max_session_permit=10\n)\nresults = await crawler.arun_many(\n    urls=[\"https://site1.com\", \"https://site2.com\",…", "filename": ""}], "chunk_position": 21, "heading_path": "With a Custom Dispatcher > With a Custom Dispatcher", "breadcrumbs": "arun_many() > With a Custom Dispatcher > With a Custom Dispatcher"}, {"id": "d0bcc3fc53add5b5", "url": "https://docs.crawl4ai.com/api/arun_many/", "page_title": "arun_many()", "page_type": "api", "page_summary": "Reference for the arun_many() function in Crawl4AI, which crawls multiple URLs concurrently or in batches, with support for dispatchers, streaming, and per-URL configurations.", "heading": "URL-Specific Configurations", "content": "Page: arun_many()\nSection: URL-Specific Configurations\n\nInstead of using one config for all URLs, provide a list of configs with `url_matcher` patterns:\n\n**URL Matching Features**:\n- **String patterns**: `\"*.pdf\"`, `\"*/blog/*\"`, `\"*python.org*\"`\n-…", "code_blocks": [{"language": "python", "code": "from crawl4ai import CrawlerRunConfig, MatchMode\nfrom crawl4ai.processors.pdf import PDFContentScrapingStrategy\nfrom crawl4ai.extraction_strategy import JsonCssExtractionStrategy\nfrom…", "filename": ""}], "chunk_position": 21, "heading_path": "URL-Specific Configurations > URL-Specific Configurations", "breadcrumbs": "arun_many() > URL-Specific Configurations > URL-Specific Configurations"}, {"id": "0d0df4108278c3c2", "url": "https://docs.crawl4ai.com/api/arun_many/", "page_title": "arun_many()", "page_type": "api", "page_summary": "Reference for the arun_many() function in Crawl4AI, which crawls multiple URLs concurrently or in batches, with support for dispatchers, streaming, and per-URL configurations.", "heading": "Return Value", "content": "Page: arun_many()\nSection: Return Value\n\nReturns a **`RunManyReturn`** object which contains either a **list** of [`CrawlResult`](../crawl-result/) objects, or an **async generator** if streaming is enabled. You can iterate to check…", "code_blocks": [], "chunk_position": 21, "heading_path": "Return Value > Return Value", "breadcrumbs": "arun_many() > Return Value > Return Value"}, {"id": "4704e7b447d29121", "url": "https://docs.crawl4ai.com/api/arun_many/", "page_title": "arun_many()", "page_type": "api", "page_summary": "Reference for the arun_many() function in Crawl4AI, which crawls multiple URLs concurrently or in batches, with support for dispatchers, streaming, and per-URL configurations.", "heading": "Dispatcher Reference", "content": "Page: arun_many()\nSection: Dispatcher Reference\n\n- **`MemoryAdaptiveDispatcher`** : Dynamically manages concurrency based on system memory usage.\n- **`SemaphoreDispatcher`** : Fixed concurrency limit, simpler but less adaptive.\n\nFor advanced usage…", "code_blocks": [], "chunk_position": 21, "heading_path": "Dispatcher Reference > Dispatcher Reference", "breadcrumbs": "arun_many() > Dispatcher Reference > Dispatcher Reference"}, {"id": "1a90498faf32e2b3", "url": "https://docs.crawl4ai.com/api/arun_many/", "page_title": "arun_many()", "page_type": "api", "page_summary": "Reference for the arun_many() function in Crawl4AI, which crawls multiple URLs concurrently or in batches, with support for dispatchers, streaming, and per-URL configurations.", "heading": "Common Pitfalls", "content": "Page: arun_many()\nSection: Common Pitfalls\n\n1. **Large Lists** : If you pass thousands of URLs, be mindful of memory or rate-limits. A dispatcher can help.\n2. **Session Reuse** : If you need specialized logins or persistent contexts, ensure…", "code_blocks": [], "chunk_position": 21, "heading_path": "Common Pitfalls > Common Pitfalls", "breadcrumbs": "arun_many() > Common Pitfalls > Common Pitfalls"}, {"id": "7163f26d970e7829", "url": "https://docs.crawl4ai.com/api/arun_many/", "page_title": "arun_many()", "page_type": "api", "page_summary": "Reference for the arun_many() function in Crawl4AI, which crawls multiple URLs concurrently or in batches, with support for dispatchers, streaming, and per-URL configurations.", "heading": "Conclusion", "content": "Page: arun_many()\nSection: Conclusion\n\nUse `arun_many()` when you want to **crawl multiple URLs** simultaneously or in controlled parallel tasks. If you need advanced concurrency features (like memory-based adaptive throttling or complex…", "code_blocks": [], "chunk_position": 21, "heading_path": "Conclusion > Conclusion", "breadcrumbs": "arun_many() > Conclusion > Conclusion"}, {"id": "229e8da580229dcd", "url": "https://docs.crawl4ai.com/api/async-webcrawler/", "page_title": "AsyncWebCrawler - Crawl4AI Documentation (v0.9.x)", "page_type": "reference", "page_summary": "Extraction fallback content.", "heading": "AsyncWebCrawler - Crawl4AI Documentation (v0.9.x)", "content": "Page: AsyncWebCrawler - Crawl4AI Documentation (v0.9.x)\nSection: AsyncWebCrawler - Crawl4AI Documentation (v0.9.x)\n\n\n\n\n##### AsyncWebCrawler\n\n\n\n\nThe  **`AsyncWebCrawler`**  is the core class for asynchronous web crawling in Crawl4AI. You typically create it  **once** , optionally customize it with a  **`BrowserConfig`**  (e.g., headless, user agent), then  **run**  multiple  **`arun()`**  calls with different  **`CrawlerRunConfig`**  objects.\n\n\n\n\n **Recommended usage** :\n\n\n\n\n1.  **Create**  a `BrowserConfig` for global browser settings.  \n\n\n\n\n2.  **Instantiate**  `AsyncWebCrawler(config=browser_config)`.  \n\n\n\n\n3.  **Use**  the crawler in an async context manager (`async with`) or manage start/close manually.  \n\n\n\n\n4.  **Call**  `arun(url, config=crawler_run_config)` for each page you want.\n\n\n\n\n\n##### 1. Constructor Overview\n\n\n\n\n\n **Notes** :\n\n\n\n\n\n\n- **Legacy**  parameters like `always_bypass_cache` remain for backward compatibility, but prefer to set  **caching**  in `CrawlerRunConfig`.\n\n\n\n\n\n\n##### 2. Lifecycle: Start/Close or Context Manager\n\n\n\n\n###### 2.1 Context Manager (Recommended)\n\n\n\n\n\nWhen the `async with` block ends, the crawler cleans up (closes the browser, etc.).\n\n\n\n\n###### 2.2 Manual Start & Close\n\n\n\n\n\nUse this style if you have a  **long-running**  application or need full control of the crawler’s lifecycle.\n\n\n\n\n\n##### 3. Primary Method: `arun()`\n\n\n\n\n\n###### 3.1 New Approach\n\n\n\n\nYou pass a `CrawlerRunConfig` object that sets up everything about a crawl—content filtering, caching, session reuse, JS code, screenshots, etc.\n\n\n\n\n\n###### 3.2 Legacy Parameters Still Accepted\n\n\n\n\nFor  **backward**  compatibility, `arun()` can still accept direct arguments like `css_selector=...`, `word_count_threshold=...`, etc., but we strongly advise migrating them into a  **`CrawlerRunConfig`** .\n\n\n\n\n\n##### 4. Batch Processing: `arun_many()`\n\n\n\n\n\n###### 4.1 Resource-Aware Crawling\n\n\n\n\nThe `arun_many()` method now uses an intelligent dispatcher that:\n\n\n\n\n\n\n- Monitors system memory usage\n\n- Implements adaptive rate limiting\n\n- Provides detailed progress monitoring\n\n- Manages concurrent crawls efficiently\n\n\n\n\n\n###### 4.2 Example Usage\n\n\n\n\nCheck page [Multi-url Crawling](../../advanced/multi-url-crawling/) for a detailed example of how to use `arun_many()`.\n\n\n\n\n\n **Explanation** :\n\n\n\n\n\n\n- We define a  **`BrowserConfig`**  with Firefox, no headless, and `verbose=True`.\n\n- We define a  **`CrawlerRunConfig`**  that  **bypasses cache** , uses a  **CSS**  extraction schema, has a `word_count_threshold=15`, etc.\n\n- We pass them to `AsyncWebCrawler(config=...)` and `arun(url=..., config=...)`.\n\n\n\n\n\n\n##### 7. Best Practices & Migration Notes\n\n\n\n\n1.  **Use**  `BrowserConfig` for  **global**  settings about the browser’s environment.  \n2.  **Use**  `CrawlerRunConfig` for  **per-crawl**  logic (caching, content filtering, extraction strategies, wait conditions).  \n3.  **Avoid**  legacy parameters like `css_selector` or `word_count_threshold` directly in `arun()`. Instead:\n\n\n\n\n\n4.  **Context Manager**  usage is simplest unless you want a persistent crawler across many calls.\n\n\n\n\n\n##### 8. Summary\n\n\n\n\n **AsyncWebCrawler**  is your entry point to asynchronous crawling:\n\n\n\n\n\n\n- **Constructor**  accepts  **`BrowserConfig`**  (or defaults).\n\n- **`arun(url, config=CrawlerRunConfig)`**  is the main method for single-page crawls.\n\n- **`arun_many(urls, config=CrawlerRunConfig)`**  handles concurrency across multiple URLs.\n\n- For advanced lifecycle control, use `start()` and `close()` explicitly.\n\n\n\n\n\n **Migration** :  \n\n\n\n\n\n\n- If you used `AsyncWebCrawler(browser_type=\"chromium\", css_selector=\"...\")`, move browser settings to `BrowserConfig(...)` and content/crawl logic to `CrawlerRunConfig(...)`.\n\n\n\n\n\nThis modular approach ensures your code is  **clean** ,  **scalable** , and  **easy to maintain** . For any advanced or rarely used parameters, see the [BrowserConfig docs](../parameters/).\n\n\n", "code_blocks": [{"language": "python", "code": "class AsyncWebCrawler:\n    def __init__(\n        self,\n        crawler_strategy: Optional[AsyncCrawlerStrategy] = None,\n        config: Optional[BrowserConfig] = None,\n        always_bypass_cache: bool = False,           # deprecated\n        always_by_pass_cache: Optional[bool] = None, # also deprecated\n        base_directory: str = ...,\n        thread_safe: bool = False,\n        **kwargs,\n    ):\n        \"\"\"\n        Create an AsyncWebCrawler instance.\n\n        Args:\n            crawler_strategy: \n                (Advanced) Provide a custom crawler strategy if needed.\n            config: \n                A BrowserConfig object specifying how the browser is set up.\n            always_bypass_cache: \n                (Deprecated) Use CrawlerRunConfig.cache_mode instead.\n            base_directory:     \n                Folder for storing caches/logs (if relevant).\n            thread_safe: \n                If True, attempts some concurrency safeguards. Usually False.\n            **kwargs: \n                Additional legacy or debugging parameters.\n        \"\"\"\n    )\n\n### Typical Initialization\n\n```python\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig\n\nbrowser_cfg = BrowserConfig(\n    browser_type=\"chromium\",\n    headless=True,\n    verbose=True\n)\n\ncrawler = AsyncWebCrawler(config=browser_cfg)", "filename": ""}, {"language": "csharp", "code": "async with AsyncWebCrawler(config=browser_cfg) as crawler:\n    result = await crawler.arun(\"https://example.com\")\n    # The crawler automatically starts/closes resources", "filename": ""}, {"language": "csharp", "code": "crawler = AsyncWebCrawler(config=browser_cfg)\nawait crawler.start()\n\nresult1 = await crawler.arun(\"https://example.com\")\nresult2 = await crawler.arun(\"https://another.com\")\n\nawait crawler.close()", "filename": ""}, {"language": "python", "code": "async def arun(\n    self,\n    url: str,\n    config: Optional[CrawlerRunConfig] = None,\n    # Legacy parameters for backward compatibility...\n) -> RunManyReturn:\n    ...", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import CrawlerRunConfig, CacheMode\n\nrun_cfg = CrawlerRunConfig(\n    cache_mode=CacheMode.BYPASS,\n    css_selector=\"main.article\",\n    word_count_threshold=10,\n    screenshot=True\n)\n\nasync with AsyncWebCrawler(config=browser_cfg) as crawler:\n    result = await crawler.arun(\"https://example.com/news\", config=run_cfg)\n    print(\"Crawled HTML length:\", len(result.cleaned_html))\n    if result.screenshot:\n        print(\"Screenshot base64 length:\", len(result.screenshot))", "filename": ""}, {"language": "python", "code": "async def arun_many(\n    self,\n    urls: List[str],\n    config: Optional[CrawlerRunConfig] = None,\n    # Legacy parameters maintained for backwards compatibility...\n) -> RunManyReturn:\n    \"\"\"\n    Process multiple URLs with intelligent rate limiting and resource monitoring.\n    \"\"\"", "filename": ""}, {"language": "python", "code": "### 4.3 Key Features\n\n1. **Rate Limiting**\n\n   - Automatic delay between requests\n   - Exponential backoff on rate limit detection\n   - Domain-specific rate limiting\n   - Configurable retry strategy\n\n2. **Resource Monitoring**\n\n   - Memory usage tracking\n   - Adaptive concurrency based on system load\n   - Automatic pausing when resources are constrained\n\n3. **Progress Monitoring**\n\n   - Detailed or aggregated progress display\n   - Real-time status updates\n   - Memory usage statistics\n\n4. **Error Handling**\n\n   - Graceful handling of rate limits\n   - Automatic retries with backoff\n   - Detailed error reporting\n\n---\n\n## 5. `CrawlResult` Output\n\nEach `arun()` returns a **`CrawlResult`** containing:\n\n- `url`: Final URL (if redirected).\n- `html`: Original HTML.\n- `cleaned_html`: Sanitized HTML.\n- `markdown_v2`: Removed in v0.5. Accessing it raises `AttributeError`; use `markdown`.\n- `extracted_content`: If an extraction strategy was used (JSON for CSS/LLM strategies).\n- `screenshot`, `pdf`: If screenshots/PDF requested.\n- `media`, `links`: Information about discovered images/links.\n- `success`, `error_message`: Status info.\n\nFor details, see [CrawlResult doc](./crawl-result.md).\n\n---\n\n## 6. Quick Example\n\nBelow is an example hooking it all together:\n\n```python\nimport asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode\nfrom crawl4ai import JsonCssExtractionStrategy\nimport json\n\nasync def main():\n    # 1. Browser config\n    browser_cfg = BrowserConfig(\n        browser_type=\"firefox\",\n        headless=False,\n        verbose=True\n    )\n\n    # 2. Run config\n    schema = {\n        \"name\": \"Articles\",\n        \"baseSelector\": \"article.post\",\n        \"fields\": [\n            {\n                \"name\": \"title\", \n                \"selector\": \"h2\", \n                \"type\": \"text\"\n            },\n            {\n                \"name\": \"url\", \n                \"selector\": \"a\", \n                \"type\": \"attribute\", \n                \"attribute\": \"href\"\n            }\n        ]\n    }\n\n    run_cfg = CrawlerRunConfig(\n        cache_mode=CacheMode.BYPASS,\n        extraction_strategy=JsonCssExtractionStrategy(schema),\n        word_count_threshold=15,\n        remove_overlay_elements=True,\n        wait_for=\"css:.post\"  # Wait for posts to appear\n    )\n\n    async with AsyncWebCrawler(config=browser_cfg) as crawler:\n        result = await crawler.arun(\n            url=\"https://example.com/blog\",\n            config=run_cfg\n        )\n\n        if result.success:\n            print(\"Cleaned HTML length:\", len(result.cleaned_html))\n            if result.extracted_content:\n                articles = json.loads(result.extracted_content)\n                print(\"Extracted articles:\", articles[:2])\n        else:\n            print(\"Error:\", result.error_message)\n\nasyncio.run(main())", "filename": ""}, {"language": "ini", "code": "run_cfg = CrawlerRunConfig(css_selector=\".main-content\", word_count_threshold=20)\nresult = await crawler.arun(url=\"...\", config=run_cfg)", "filename": ""}], "chunk_position": 22, "heading_path": "AsyncWebCrawler - Crawl4AI Documentation (v0.9.x) > AsyncWebCrawler - Crawl4AI Documentation (v0.9.x)", "breadcrumbs": "AsyncWebCrawler - Crawl4AI Documentation (v0.9.x) > AsyncWebCrawler - Crawl4AI Documentation (v0.9.x) > AsyncWebCrawler - Crawl4AI Documentation (v0.9.x)"}, {"id": "b883169e74f8268c", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "Command Categories", "content": "Page: C4A-Script API Reference\nSection: Command Categories\n\nComplete reference for all C4A-Script commands, syntax, and advanced features.", "code_blocks": [], "chunk_position": 23, "heading_path": "Command Categories > Command Categories", "breadcrumbs": "C4A-Script API Reference > Command Categories > Command Categories"}, {"id": "dadc7efa0c2d4bb7", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "🧭 Navigation Commands", "content": "Page: C4A-Script API Reference\nSection: 🧭 Navigation Commands\n\nNavigate between pages and manage browser history.", "code_blocks": [], "chunk_position": 23, "heading_path": "🧭 Navigation Commands > 🧭 Navigation Commands", "breadcrumbs": "C4A-Script API Reference > 🧭 Navigation Commands > 🧭 Navigation Commands"}, {"id": "602a29cd4625c4ed", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`GO <url>`", "content": "Page: C4A-Script API Reference\nSection: `GO <url>`\n\nNavigate to a specific URL.\n\n**Syntax:**\n\n```\nGO <url>\n```\n\n**Parameters:**\n- `url` - Target URL (string)\n\n**Examples:**\n\n```\nGO https://example.com\nGO https://api.example.com/login\nGO…", "code_blocks": [{"language": "", "code": "GO <url>", "filename": ""}, {"language": "", "code": "GO https://example.com\nGO https://api.example.com/login\nGO /relative/path", "filename": ""}], "chunk_position": 23, "heading_path": "`GO <url>` > `GO <url>`", "breadcrumbs": "C4A-Script API Reference > `GO <url>` > `GO <url>`"}, {"id": "0877d96c481128a4", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`RELOAD`", "content": "Page: C4A-Script API Reference\nSection: `RELOAD`\n\nRefresh the current page.\n\n**Syntax:**\n\n```\nRELOAD\n```\n\n**Examples:**\n\n```\nRELOAD\n```\n\n**Notes:**\n- Equivalent to pressing F5 or clicking browser refresh\n- Waits for page reload to complete\n-…", "code_blocks": [{"language": "", "code": "RELOAD", "filename": ""}, {"language": "", "code": "RELOAD", "filename": ""}], "chunk_position": 23, "heading_path": "`RELOAD` > `RELOAD`", "breadcrumbs": "C4A-Script API Reference > `RELOAD` > `RELOAD`"}, {"id": "c83269db8ef3cfe9", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`BACK`", "content": "Page: C4A-Script API Reference\nSection: `BACK`\n\nNavigate back in browser history.\n\n**Syntax:**\n\n```\nBACK\n```\n\n**Examples:**\n\n```\nBACK\n```\n\n**Notes:**\n- Equivalent to clicking browser back button\n- Does nothing if no previous page exists\n- Waits…", "code_blocks": [{"language": "", "code": "BACK", "filename": ""}, {"language": "", "code": "BACK", "filename": ""}], "chunk_position": 23, "heading_path": "`BACK` > `BACK`", "breadcrumbs": "C4A-Script API Reference > `BACK` > `BACK`"}, {"id": "dd4896d8f10e2ac9", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`FORWARD`", "content": "Page: C4A-Script API Reference\nSection: `FORWARD`\n\nNavigate forward in browser history.\n\n**Syntax:**\n\n```\nFORWARD\n```\n\n**Examples:**\n\n```\nFORWARD\n```\n\n**Notes:**\n- Equivalent to clicking browser forward button\n- Does nothing if no next page exists\n-…", "code_blocks": [{"language": "", "code": "FORWARD", "filename": ""}, {"language": "", "code": "FORWARD", "filename": ""}], "chunk_position": 23, "heading_path": "`FORWARD` > `FORWARD`", "breadcrumbs": "C4A-Script API Reference > `FORWARD` > `FORWARD`"}, {"id": "cefc3441b2b8e799", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "⏱️ Wait Commands", "content": "Page: C4A-Script API Reference\nSection: ⏱️ Wait Commands\n\nControl timing and synchronization with page elements.", "code_blocks": [], "chunk_position": 23, "heading_path": "⏱️ Wait Commands > ⏱️ Wait Commands", "breadcrumbs": "C4A-Script API Reference > ⏱️ Wait Commands > ⏱️ Wait Commands"}, {"id": "a861d78816eac2fe", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`WAIT <time>`", "content": "Page: C4A-Script API Reference\nSection: `WAIT <time>`\n\nWait for a specified number of seconds.\n\n**Syntax:**\n\n```\nWAIT <seconds>\n```\n\n**Parameters:**\n- `seconds` - Number of seconds to wait (number)\n\n**Examples:**\n\n```\nWAIT 3\nWAIT 1.5\nWAIT…", "code_blocks": [{"language": "", "code": "WAIT <seconds>", "filename": ""}, {"language": "", "code": "WAIT 3\nWAIT 1.5\nWAIT 10", "filename": ""}], "chunk_position": 23, "heading_path": "`WAIT <time>` > `WAIT <time>`", "breadcrumbs": "C4A-Script API Reference > `WAIT <time>` > `WAIT <time>`"}, {"id": "c27cad14ba99d531", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`WAIT <selector> <timeout>`", "content": "Page: C4A-Script API Reference\nSection: `WAIT <selector> <timeout>`\n\nWait for an element to appear on the page.\n\n**Syntax:**\n\n```\nWAIT `<selector>` <timeout>\n```\n\n**Parameters:**\n- `selector` - CSS selector for the element (string in backticks)\n- `timeout` - Maximum…", "code_blocks": [{"language": "", "code": "WAIT `<selector>` <timeout>", "filename": ""}, {"language": "", "code": "WAIT `#content` 10\nWAIT `.loading-spinner` 5\nWAIT `button[type=\"submit\"]` 15\nWAIT `.results .item:first-child` 8", "filename": ""}], "chunk_position": 23, "heading_path": "`WAIT <selector> <timeout>` > `WAIT <selector> <timeout>`", "breadcrumbs": "C4A-Script API Reference > `WAIT <selector> <timeout>` > `WAIT <selector> <timeout>`"}, {"id": "1286888800ec9d66", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`WAIT \"<text>\" <timeout>`", "content": "Page: C4A-Script API Reference\nSection: `WAIT \"<text>\" <timeout>`\n\nWait for specific text to appear anywhere on the page.\n\n**Syntax:**\n\n```\nWAIT \"<text>\" <timeout>\n```\n\n**Parameters:**\n- `text` - Text content to wait for (string in quotes)\n- `timeout` - Maximum…", "code_blocks": [{"language": "", "code": "WAIT \"<text>\" <timeout>", "filename": ""}, {"language": "", "code": "WAIT \"Loading complete\" 10\nWAIT \"Welcome back\" 5\nWAIT \"Search results\" 15", "filename": ""}], "chunk_position": 23, "heading_path": "`WAIT \"<text>\" <timeout>` > `WAIT \"<text>\" <timeout>`", "breadcrumbs": "C4A-Script API Reference > `WAIT \"<text>\" <timeout>` > `WAIT \"<text>\" <timeout>`"}, {"id": "4aca409091596b93", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`CLICK <selector>`", "content": "Page: C4A-Script API Reference\nSection: `CLICK <selector>`\n\nClick on an element specified by CSS selector.\n\n**Syntax:**\n\n```\nCLICK `<selector>`\n```\n\n**Parameters:**\n- `selector` - CSS selector for the element (string in backticks)\n\n**Examples:**\n\n```\nCLICK…", "code_blocks": [{"language": "", "code": "CLICK `<selector>`", "filename": ""}, {"language": "", "code": "CLICK `#submit-button`\nCLICK `.menu-item:first-child`\nCLICK `button[data-action=\"save\"]`\nCLICK `a[href=\"/dashboard\"]`", "filename": ""}], "chunk_position": 23, "heading_path": "`CLICK <selector>` > `CLICK <selector>`", "breadcrumbs": "C4A-Script API Reference > `CLICK <selector>` > `CLICK <selector>`"}, {"id": "83bc30147a0f0565", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`CLICK <x> <y>`", "content": "Page: C4A-Script API Reference\nSection: `CLICK <x> <y>`\n\nClick at specific coordinates on the page.\n\n**Syntax:**\n\n```\nCLICK <x> <y>\n```\n\n**Parameters:**\n- `x` - X coordinate in pixels (number)\n- `y` - Y coordinate in pixels…", "code_blocks": [{"language": "", "code": "CLICK <x> <y>", "filename": ""}, {"language": "", "code": "CLICK 100 200\nCLICK 500 300\nCLICK 0 0", "filename": ""}], "chunk_position": 23, "heading_path": "`CLICK <x> <y>` > `CLICK <x> <y>`", "breadcrumbs": "C4A-Script API Reference > `CLICK <x> <y>` > `CLICK <x> <y>`"}, {"id": "2fc31206fb677902", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`DOUBLE_CLICK <selector>`", "content": "Page: C4A-Script API Reference\nSection: `DOUBLE_CLICK <selector>`\n\nDouble-click on an element.\n\n**Syntax:**\n\n```\nDOUBLE_CLICK `<selector>`\n```\n\n**Parameters:**\n- `selector` - CSS selector for the element (string in backticks)\n\n**Examples:**\n\n```\nDOUBLE_CLICK…", "code_blocks": [{"language": "", "code": "DOUBLE_CLICK `<selector>`", "filename": ""}, {"language": "", "code": "DOUBLE_CLICK `.file-icon`\nDOUBLE_CLICK `#editable-cell`\nDOUBLE_CLICK `.expandable-item`", "filename": ""}], "chunk_position": 23, "heading_path": "`DOUBLE_CLICK <selector>` > `DOUBLE_CLICK <selector>`", "breadcrumbs": "C4A-Script API Reference > `DOUBLE_CLICK <selector>` > `DOUBLE_CLICK <selector>`"}, {"id": "5aa21990290eae5a", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`RIGHT_CLICK <selector>`", "content": "Page: C4A-Script API Reference\nSection: `RIGHT_CLICK <selector>`\n\nRight-click on an element to open context menu.\n\n**Syntax:**\n\n```\nRIGHT_CLICK `<selector>`\n```\n\n**Parameters:**\n- `selector` - CSS selector for the element (string in…", "code_blocks": [{"language": "", "code": "RIGHT_CLICK `<selector>`", "filename": ""}, {"language": "", "code": "RIGHT_CLICK `#context-target`\nRIGHT_CLICK `.menu-trigger`\nRIGHT_CLICK `img.thumbnail`", "filename": ""}], "chunk_position": 23, "heading_path": "`RIGHT_CLICK <selector>` > `RIGHT_CLICK <selector>`", "breadcrumbs": "C4A-Script API Reference > `RIGHT_CLICK <selector>` > `RIGHT_CLICK <selector>`"}, {"id": "f94fdf5cdf461b87", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`SCROLL <direction> <amount>`", "content": "Page: C4A-Script API Reference\nSection: `SCROLL <direction> <amount>`\n\nScroll the page in a specified direction.\n\n**Syntax:**\n\n```\nSCROLL <direction> <amount>\n```\n\n**Parameters:**\n- `direction` - Direction to scroll: `UP`, `DOWN`, `LEFT`, `RIGHT`\n- `amount` - Number of…", "code_blocks": [{"language": "", "code": "SCROLL <direction> <amount>", "filename": ""}, {"language": "", "code": "SCROLL DOWN 500\nSCROLL UP 200\nSCROLL LEFT 100\nSCROLL RIGHT 300", "filename": ""}], "chunk_position": 23, "heading_path": "`SCROLL <direction> <amount>` > `SCROLL <direction> <amount>`", "breadcrumbs": "C4A-Script API Reference > `SCROLL <direction> <amount>` > `SCROLL <direction> <amount>`"}, {"id": "87fde960327f5ce1", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`MOVE <x> <y>`", "content": "Page: C4A-Script API Reference\nSection: `MOVE <x> <y>`\n\nMove mouse cursor to specific coordinates.\n\n**Syntax:**\n\n```\nMOVE <x> <y>\n```\n\n**Parameters:**\n- `x` - X coordinate in pixels (number)\n- `y` - Y coordinate in pixels (number)\n\n**Examples:**\n\n```\nMOVE…", "code_blocks": [{"language": "", "code": "MOVE <x> <y>", "filename": ""}, {"language": "", "code": "MOVE 200 100\nMOVE 500 400", "filename": ""}], "chunk_position": 23, "heading_path": "`MOVE <x> <y>` > `MOVE <x> <y>`", "breadcrumbs": "C4A-Script API Reference > `MOVE <x> <y>` > `MOVE <x> <y>`"}, {"id": "459168556e48ca38", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`DRAG <x1> <y1> <x2> <y2>`", "content": "Page: C4A-Script API Reference\nSection: `DRAG <x1> <y1> <x2> <y2>`\n\nDrag from one point to another.\n\n**Syntax:**\n\n```\nDRAG <x1> <y1> <x2> <y2>\n```\n\n**Parameters:**\n- `x1`, `y1` - Starting coordinates (numbers)\n- `x2`, `y2` - Ending coordinates…", "code_blocks": [{"language": "", "code": "DRAG <x1> <y1> <x2> <y2>", "filename": ""}, {"language": "", "code": "DRAG 100 100 500 300\nDRAG 0 200 400 200", "filename": ""}], "chunk_position": 23, "heading_path": "`DRAG <x1> <y1> <x2> <y2>` > `DRAG <x1> <y1> <x2> <y2>`", "breadcrumbs": "C4A-Script API Reference > `DRAG <x1> <y1> <x2> <y2>` > `DRAG <x1> <y1> <x2> <y2>`"}, {"id": "2a244d110508b8fa", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`TYPE \"<text>\"`", "content": "Page: C4A-Script API Reference\nSection: `TYPE \"<text>\"`\n\nType text into the currently focused element.\n\n**Syntax:**\n\n```\nTYPE \"<text>\"\n```\n\n**Parameters:**\n- `text` - Text to type (string in quotes)\n\n**Examples:**\n\n```\nTYPE \"Hello, World!\"\nTYPE…", "code_blocks": [{"language": "", "code": "TYPE \"<text>\"", "filename": ""}, {"language": "", "code": "TYPE \"Hello, World!\"\nTYPE \"user@example.com\"\nTYPE \"Password123!\"", "filename": ""}], "chunk_position": 23, "heading_path": "`TYPE \"<text>\"` > `TYPE \"<text>\"`", "breadcrumbs": "C4A-Script API Reference > `TYPE \"<text>\"` > `TYPE \"<text>\"`"}, {"id": "852bd4fbf1c687dd", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`TYPE $<variable>`", "content": "Page: C4A-Script API Reference\nSection: `TYPE $<variable>`\n\nType the value of a variable.\n\n**Syntax:**\n\n```\nTYPE $<variable>\n```\n\n**Parameters:**\n- `variable` - Variable name (without quotes)\n\n**Examples:**\n\n```\nSETVAR email = \"user@example.com\"\nTYPE…", "code_blocks": [{"language": "", "code": "TYPE $<variable>", "filename": ""}, {"language": "", "code": "SETVAR email = \"user@example.com\"\nTYPE $email", "filename": ""}], "chunk_position": 23, "heading_path": "`TYPE $<variable>` > `TYPE $<variable>`", "breadcrumbs": "C4A-Script API Reference > `TYPE $<variable>` > `TYPE $<variable>`"}, {"id": "ddd1bc1b3e5dc273", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`PRESS <key>`", "content": "Page: C4A-Script API Reference\nSection: `PRESS <key>`\n\nPress and release a special key.\n\n**Syntax:**\n\n```\nPRESS <key>\n```\n\n**Parameters:**\n- `key` - Key name (see supported keys below)\n\n**Supported Keys:**\n- `Tab`, `Enter`, `Escape`, `Space`\n- `ArrowUp`,…", "code_blocks": [{"language": "", "code": "PRESS <key>", "filename": ""}, {"language": "", "code": "PRESS Tab\nPRESS Enter\nPRESS Escape\nPRESS ArrowDown", "filename": ""}], "chunk_position": 23, "heading_path": "`PRESS <key>` > `PRESS <key>`", "breadcrumbs": "C4A-Script API Reference > `PRESS <key>` > `PRESS <key>`"}, {"id": "d1b6635f6cd96ae4", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`KEY_DOWN <key>`", "content": "Page: C4A-Script API Reference\nSection: `KEY_DOWN <key>`\n\nHold down a modifier key.\n\n**Syntax:**\n\n```\nKEY_DOWN <key>\n```\n\n**Parameters:**\n- `key` - Modifier key: `Shift`, `Control`, `Alt`, `Meta`\n\n**Examples:**\n\n```\nKEY_DOWN Shift\nKEY_DOWN…", "code_blocks": [{"language": "", "code": "KEY_DOWN <key>", "filename": ""}, {"language": "", "code": "KEY_DOWN Shift\nKEY_DOWN Control", "filename": ""}], "chunk_position": 23, "heading_path": "`KEY_DOWN <key>` > `KEY_DOWN <key>`", "breadcrumbs": "C4A-Script API Reference > `KEY_DOWN <key>` > `KEY_DOWN <key>`"}, {"id": "710c6e3aadee2edc", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`KEY_UP <key>`", "content": "Page: C4A-Script API Reference\nSection: `KEY_UP <key>`\n\nRelease a modifier key.\n\n**Syntax:**\n\n```\nKEY_UP <key>\n```\n\n**Parameters:**\n- `key` - Modifier key: `Shift`, `Control`, `Alt`, `Meta`\n\n**Examples:**\n\n```\nKEY_UP Shift\nKEY_UP Control\n```\n\n**Notes:**\n-…", "code_blocks": [{"language": "", "code": "KEY_UP <key>", "filename": ""}, {"language": "", "code": "KEY_UP Shift\nKEY_UP Control", "filename": ""}], "chunk_position": 23, "heading_path": "`KEY_UP <key>` > `KEY_UP <key>`", "breadcrumbs": "C4A-Script API Reference > `KEY_UP <key>` > `KEY_UP <key>`"}, {"id": "a46d60936ff62fdd", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`CLEAR <selector>`", "content": "Page: C4A-Script API Reference\nSection: `CLEAR <selector>`\n\nClear the content of an input field.\n\n**Syntax:**\n\n```\nCLEAR `<selector>`\n```\n\n**Parameters:**\n- `selector` - CSS selector for input element (string in backticks)\n\n**Examples:**\n\n```\nCLEAR…", "code_blocks": [{"language": "", "code": "CLEAR `<selector>`", "filename": ""}, {"language": "", "code": "CLEAR `#search-box`\nCLEAR `input[name=\"email\"]`\nCLEAR `.form-input:first-child`", "filename": ""}], "chunk_position": 23, "heading_path": "`CLEAR <selector>` > `CLEAR <selector>`", "breadcrumbs": "C4A-Script API Reference > `CLEAR <selector>` > `CLEAR <selector>`"}, {"id": "1611481017797534", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`SET <selector> \"<value>\"`", "content": "Page: C4A-Script API Reference\nSection: `SET <selector> \"<value>\"`\n\nSet the value of an input field directly.\n\n**Syntax:**\n\n```\nSET `<selector>` \"<value>\"\n```\n\n**Parameters:**\n- `selector` - CSS selector for input element (string in backticks)\n- `value` - Value to…", "code_blocks": [{"language": "", "code": "SET `<selector>` \"<value>\"", "filename": ""}, {"language": "", "code": "SET `#email` \"user@example.com\"\nSET `#age` \"25\"\nSET `textarea#message` \"Hello, this is a test message.\"", "filename": ""}], "chunk_position": 23, "heading_path": "`SET <selector> \"<value>\"` > `SET <selector> \"<value>\"`", "breadcrumbs": "C4A-Script API Reference > `SET <selector> \"<value>\"` > `SET <selector> \"<value>\"`"}, {"id": "bf717e4d2976ee83", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`IF (EXISTS <selector>) THEN <command>`", "content": "Page: C4A-Script API Reference\nSection: `IF (EXISTS <selector>) THEN <command>`\n\nExecute command if element exists.\n\n**Syntax:**\n\n```\nIF (EXISTS `<selector>`) THEN <command>\n```\n\n**Parameters:**\n- `selector` - CSS selector to check (string in backticks)\n- `command` - Command to…", "code_blocks": [{"language": "", "code": "IF (EXISTS `<selector>`) THEN <command>", "filename": ""}, {"language": "", "code": "IF (EXISTS `.cookie-banner`) THEN CLICK `.accept-cookies`\nIF (EXISTS `#popup-modal`) THEN CLICK `.close-button`\nIF (EXISTS `.error-message`) THEN RELOAD", "filename": ""}], "chunk_position": 23, "heading_path": "`IF (EXISTS <selector>) THEN <command>` > `IF (EXISTS <selector>) THEN <command>`", "breadcrumbs": "C4A-Script API Reference > `IF (EXISTS <selector>) THEN <command>` > `IF (EXISTS <selector>) THEN <command>`"}, {"id": "a5a40c1c675f48c8", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`IF (EXISTS <selector>) THEN <command> ELSE <command>`", "content": "Page: C4A-Script API Reference\nSection: `IF (EXISTS <selector>) THEN <command> ELSE <command>`\n\nExecute command based on element existence.\n\n**Syntax:**\n\n```\nIF (EXISTS `<selector>`) THEN <command> ELSE <command>\n```\n\n**Parameters:**\n- `selector` - CSS selector to check (string in backticks)\n-…", "code_blocks": [{"language": "", "code": "IF (EXISTS `<selector>`) THEN <command> ELSE <command>", "filename": ""}, {"language": "", "code": "IF (EXISTS `.user-menu`) THEN CLICK `.logout` ELSE CLICK `.login`\nIF (EXISTS `.loading`) THEN WAIT 5 ELSE CLICK `#continue`", "filename": ""}], "chunk_position": 23, "heading_path": "`IF (EXISTS <selector>) THEN <command> ELSE <command>` > `IF (EXISTS <selector>) THEN <command> ELSE <command>`", "breadcrumbs": "C4A-Script API Reference > `IF (EXISTS <selector>) THEN <command> ELSE <command>` > `IF (EXISTS <selector>) THEN <command> ELSE <command>`"}, {"id": "6b0039f64c19ca07", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`IF (NOT EXISTS <selector>) THEN <command>`", "content": "Page: C4A-Script API Reference\nSection: `IF (NOT EXISTS <selector>) THEN <command>`\n\nExecute command if element does not exist.\n\n**Syntax:**\n\n```\nIF (NOT EXISTS `<selector>`) THEN <command>\n```\n\n**Parameters:**\n- `selector` - CSS selector to check (string in backticks)\n- `command` -…", "code_blocks": [{"language": "", "code": "IF (NOT EXISTS `<selector>`) THEN <command>", "filename": ""}, {"language": "", "code": "IF (NOT EXISTS `.logged-in`) THEN GO /login\nIF (NOT EXISTS `.results`) THEN CLICK `#search-button`", "filename": ""}], "chunk_position": 23, "heading_path": "`IF (NOT EXISTS <selector>) THEN <command>` > `IF (NOT EXISTS <selector>) THEN <command>`", "breadcrumbs": "C4A-Script API Reference > `IF (NOT EXISTS <selector>) THEN <command>` > `IF (NOT EXISTS <selector>) THEN <command>`"}, {"id": "02def72ab064ca4c", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`IF (<javascript>) THEN <command>`", "content": "Page: C4A-Script API Reference\nSection: `IF (<javascript>) THEN <command>`\n\nExecute command based on JavaScript condition.\n\n**Syntax:**\n\n```\nIF (`<javascript>`) THEN <command>\n```\n\n**Parameters:**\n- `javascript` - JavaScript expression that returns boolean (string in…", "code_blocks": [{"language": "", "code": "IF (`<javascript>`) THEN <command>", "filename": ""}, {"language": "", "code": "IF (`window.innerWidth < 768`) THEN CLICK `.mobile-menu`\nIF (`document.readyState === \"complete\"`) THEN CLICK `#start`\nIF (`localStorage.getItem(\"user\")`) THEN GO /dashboard", "filename": ""}], "chunk_position": 23, "heading_path": "`IF (<javascript>) THEN <command>` > `IF (<javascript>) THEN <command>`", "breadcrumbs": "C4A-Script API Reference > `IF (<javascript>) THEN <command>` > `IF (<javascript>) THEN <command>`"}, {"id": "a5bdda64e97b0fb0", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`REPEAT (<command>, <count>)`", "content": "Page: C4A-Script API Reference\nSection: `REPEAT (<command>, <count>)`\n\nRepeat a command a specific number of times.\n\n**Syntax:**\n\n```\nREPEAT (<command>, <count>)\n```\n\n**Parameters:**\n- `command` - Command to repeat\n- `count` - Number of times to repeat…", "code_blocks": [{"language": "", "code": "REPEAT (<command>, <count>)", "filename": ""}, {"language": "", "code": "REPEAT (SCROLL DOWN 300, 5)\nREPEAT (PRESS Tab, 3)\nREPEAT (CLICK `.load-more`, 10)", "filename": ""}], "chunk_position": 23, "heading_path": "`REPEAT (<command>, <count>)` > `REPEAT (<command>, <count>)`", "breadcrumbs": "C4A-Script API Reference > `REPEAT (<command>, <count>)` > `REPEAT (<command>, <count>)`"}, {"id": "95e9525018d1db8a", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`REPEAT (<command>, <condition>)`", "content": "Page: C4A-Script API Reference\nSection: `REPEAT (<command>, <condition>)`\n\nRepeat a command while condition is true.\n\n**Syntax:**\n\n```\nREPEAT (<command>, `<condition>`)\n```\n\n**Parameters:**\n- `command` - Command to repeat\n- `condition` - JavaScript condition to check…", "code_blocks": [{"language": "", "code": "REPEAT (<command>, `<condition>`)", "filename": ""}, {"language": "", "code": "REPEAT (SCROLL DOWN 500, `document.querySelector(\".load-more\")`)\nREPEAT (PRESS ArrowDown, `window.scrollY < document.body.scrollHeight`)", "filename": ""}], "chunk_position": 23, "heading_path": "`REPEAT (<command>, <condition>)` > `REPEAT (<command>, <condition>)`", "breadcrumbs": "C4A-Script API Reference > `REPEAT (<command>, <condition>)` > `REPEAT (<command>, <condition>)`"}, {"id": "63286d686c5afbf1", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`SETVAR <name> = \"<value>\"`", "content": "Page: C4A-Script API Reference\nSection: `SETVAR <name> = \"<value>\"`\n\nCreate or update a variable.\n\n**Syntax:**\n\n```\nSETVAR <name> = \"<value>\"\n```\n\n**Parameters:**\n- `name` - Variable name (alphanumeric, underscore)\n- `value` - Variable value (string in…", "code_blocks": [{"language": "", "code": "SETVAR <name> = \"<value>\"", "filename": ""}, {"language": "", "code": "SETVAR username = \"john@example.com\"\nSETVAR password = \"secret123\"\nSETVAR base_url = \"https://api.example.com\"\nSETVAR counter = \"0\"", "filename": ""}], "chunk_position": 23, "heading_path": "`SETVAR <name> = \"<value>\"` > `SETVAR <name> = \"<value>\"`", "breadcrumbs": "C4A-Script API Reference > `SETVAR <name> = \"<value>\"` > `SETVAR <name> = \"<value>\"`"}, {"id": "ea5c7b0f4e8deb84", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`EVAL <javascript>`", "content": "Page: C4A-Script API Reference\nSection: `EVAL <javascript>`\n\nExecute arbitrary JavaScript code.\n\n**Syntax:**\n\n```\nEVAL `<javascript>`\n```\n\n**Parameters:**\n- `javascript` - JavaScript code to execute (string in backticks)\n\n**Examples:**\n\n```\nEVAL…", "code_blocks": [{"language": "", "code": "EVAL `<javascript>`", "filename": ""}, {"language": "", "code": "EVAL `console.log(\"Script started\")`\nEVAL `window.scrollTo(0, 0)`\nEVAL `localStorage.setItem(\"test\", \"value\")`\nEVAL `document.title = \"Automated Test\"`", "filename": ""}], "chunk_position": 23, "heading_path": "`EVAL <javascript>` > `EVAL <javascript>`", "breadcrumbs": "C4A-Script API Reference > `EVAL <javascript>` > `EVAL <javascript>`"}, {"id": "09f08145de398afb", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`# <comment>`", "content": "Page: C4A-Script API Reference\nSection: `# <comment>`\n\nAdd comments to scripts for documentation.\n\n**Syntax:**\n\n```\n# <comment text>\n```\n\n**Examples:**\n\n```\n# This script logs into the application\n# Step 1: Navigate to login page\nGO /login\n\n# Step 2:…", "code_blocks": [{"language": "", "code": "# <comment text>", "filename": ""}, {"language": "", "code": "# This script logs into the application\n# Step 1: Navigate to login page\nGO /login\n\n# Step 2: Fill credentials\nTYPE \"user@example.com\"", "filename": ""}], "chunk_position": 23, "heading_path": "`# <comment>` > `# <comment>`", "breadcrumbs": "C4A-Script API Reference > `# <comment>` > `# <comment>`"}, {"id": "06a72aca117883a3", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`PROC <name> ... ENDPROC`", "content": "Page: C4A-Script API Reference\nSection: `PROC <name> ... ENDPROC`\n\nDefine a reusable procedure.\n\n**Syntax:**\n\n```\nPROC <name>\n  <commands>\nENDPROC\n```\n\n**Parameters:**\n- `name` - Procedure name (alphanumeric, underscore)\n- `commands` - Commands to include in…", "code_blocks": [{"language": "", "code": "PROC <name>\n  <commands>\nENDPROC", "filename": ""}, {"language": "", "code": "PROC login\n  CLICK `#email`\n  TYPE $email\n  CLICK `#password`\n  TYPE $password\n  CLICK `#submit`\nENDPROC\n\nPROC handle_popups\n  IF (EXISTS `.cookie-banner`) THEN CLICK `.accept`\n  IF (EXISTS…", "filename": ""}], "chunk_position": 23, "heading_path": "`PROC <name> ... ENDPROC` > `PROC <name> ... ENDPROC`", "breadcrumbs": "C4A-Script API Reference > `PROC <name> ... ENDPROC` > `PROC <name> ... ENDPROC`"}, {"id": "177eeff7472a6897", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "`<procedure_name>`", "content": "Page: C4A-Script API Reference\nSection: `<procedure_name>`\n\nCall a defined procedure.\n\n**Syntax:**\n\n```\n<procedure_name>\n```\n\n**Examples:**\n\n```\n# Define procedure first\nPROC setup\n  GO /login\n  WAIT `#form` 5\nENDPROC\n\n# Call…", "code_blocks": [{"language": "", "code": "<procedure_name>", "filename": ""}, {"language": "", "code": "# Define procedure first\nPROC setup\n  GO /login\n  WAIT `#form` 5\nENDPROC\n\n# Call procedure\nsetup\nlogin", "filename": ""}], "chunk_position": 23, "heading_path": "`<procedure_name>` > `<procedure_name>`", "breadcrumbs": "C4A-Script API Reference > `<procedure_name>` > `<procedure_name>`"}, {"id": "73fa1efca473f2ef", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "1. Always Use Waits", "content": "Page: C4A-Script API Reference\nSection: 1. Always Use Waits\n\n```\n# Bad - element might not be ready\nCLICK `#button`\n\n# Good - wait for element first\nWAIT `#button` 5\nCLICK `#button`\n```", "code_blocks": [{"language": "", "code": "# Bad - element might not be ready\nCLICK `#button`\n\n# Good - wait for element first\nWAIT `#button` 5\nCLICK `#button`", "filename": ""}], "chunk_position": 23, "heading_path": "1. Always Use Waits > 1. Always Use Waits", "breadcrumbs": "C4A-Script API Reference > 1. Always Use Waits > 1. Always Use Waits"}, {"id": "c02a3305c5925796", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "2. Handle Optional Elements", "content": "Page: C4A-Script API Reference\nSection: 2. Handle Optional Elements\n\n```\n# Check before interacting\nIF (EXISTS `.popup`) THEN CLICK `.close`\nIF (EXISTS `.cookie-banner`) THEN CLICK `.accept`\n\n# Then proceed with main flow\nCLICK `#main-action`\n```", "code_blocks": [{"language": "", "code": "# Check before interacting\nIF (EXISTS `.popup`) THEN CLICK `.close`\nIF (EXISTS `.cookie-banner`) THEN CLICK `.accept`\n\n# Then proceed with main flow\nCLICK `#main-action`", "filename": ""}], "chunk_position": 23, "heading_path": "2. Handle Optional Elements > 2. Handle Optional Elements", "breadcrumbs": "C4A-Script API Reference > 2. Handle Optional Elements > 2. Handle Optional Elements"}, {"id": "6056a8a09e0e500b", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "3. Use Descriptive Variables", "content": "Page: C4A-Script API Reference\nSection: 3. Use Descriptive Variables\n\n```\n# Set up reusable data\nSETVAR admin_email = \"admin@company.com\"\nSETVAR test_password = \"TestPass123!\"\nSETVAR staging_url = \"https://staging.example.com\"\n\n# Use throughout script\nGO $staging_url\nTYPE…\n```", "code_blocks": [{"language": "", "code": "# Set up reusable data\nSETVAR admin_email = \"admin@company.com\"\nSETVAR test_password = \"TestPass123!\"\nSETVAR staging_url = \"https://staging.example.com\"\n\n# Use throughout script\nGO $staging_url\nTYPE…", "filename": ""}], "chunk_position": 23, "heading_path": "3. Use Descriptive Variables > 3. Use Descriptive Variables", "breadcrumbs": "C4A-Script API Reference > 3. Use Descriptive Variables > 3. Use Descriptive Variables"}, {"id": "d86f55528ec84403", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "4. Add Debugging Information", "content": "Page: C4A-Script API Reference\nSection: 4. Add Debugging Information\n\n```\n# Log progress\nEVAL `console.log(\"Starting login process\")`\nGO /login\n\n# Verify page state\nIF (`document.title.includes(\"Login\")`) THEN EVAL `console.log(\"On login page\")`\n\n# Continue with login\nTYPE…\n```", "code_blocks": [{"language": "", "code": "# Log progress\nEVAL `console.log(\"Starting login process\")`\nGO /login\n\n# Verify page state\nIF (`document.title.includes(\"Login\")`) THEN EVAL `console.log(\"On login page\")`\n\n# Continue with login\nTYPE…", "filename": ""}], "chunk_position": 23, "heading_path": "4. Add Debugging Information > 4. Add Debugging Information", "breadcrumbs": "C4A-Script API Reference > 4. Add Debugging Information > 4. Add Debugging Information"}, {"id": "91bfa79f6da14b5d", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "Login Flow", "content": "Page: C4A-Script API Reference\nSection: Login Flow\n\n```\n# Complete login automation\nSETVAR email = \"user@example.com\"\nSETVAR password = \"mypassword\"\n\nGO /login\nWAIT `#login-form` 5\n\n# Handle optional cookie banner\nIF (EXISTS `.cookie-banner`) THEN CLICK…\n```", "code_blocks": [{"language": "", "code": "# Complete login automation\nSETVAR email = \"user@example.com\"\nSETVAR password = \"mypassword\"\n\nGO /login\nWAIT `#login-form` 5\n\n# Handle optional cookie banner\nIF (EXISTS `.cookie-banner`) THEN CLICK…", "filename": ""}], "chunk_position": 23, "heading_path": "Login Flow > Login Flow", "breadcrumbs": "C4A-Script API Reference > Login Flow > Login Flow"}, {"id": "0a33b91ad5780c08", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "Infinite Scroll", "content": "Page: C4A-Script API Reference\nSection: Infinite Scroll\n\n```\n# Load all content with infinite scroll\nGO /products\n\n# Scroll and load more content\nREPEAT (SCROLL DOWN 500, `document.querySelector(\".load-more\")`)\n\n# Alternative: Fixed number of scrolls\nREPEAT…\n```", "code_blocks": [{"language": "", "code": "# Load all content with infinite scroll\nGO /products\n\n# Scroll and load more content\nREPEAT (SCROLL DOWN 500, `document.querySelector(\".load-more\")`)\n\n# Alternative: Fixed number of scrolls\nREPEAT…", "filename": ""}], "chunk_position": 23, "heading_path": "Infinite Scroll > Infinite Scroll", "breadcrumbs": "C4A-Script API Reference > Infinite Scroll > Infinite Scroll"}, {"id": "905a49839444af80", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "Form Validation", "content": "Page: C4A-Script API Reference\nSection: Form Validation\n\n```\n# Handle form with validation\nSET `#email` \"invalid-email\"\nCLICK `#submit`\n\n# Check for validation error\nIF (EXISTS `.error-email`) THEN SET `#email` \"valid@example.com\"\n\n# Retry submission\nCLICK…\n```", "code_blocks": [{"language": "", "code": "# Handle form with validation\nSET `#email` \"invalid-email\"\nCLICK `#submit`\n\n# Check for validation error\nIF (EXISTS `.error-email`) THEN SET `#email` \"valid@example.com\"\n\n# Retry submission\nCLICK…", "filename": ""}], "chunk_position": 23, "heading_path": "Form Validation > Form Validation", "breadcrumbs": "C4A-Script API Reference > Form Validation > Form Validation"}, {"id": "fcf119d996c9d154", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "Multi-step Process", "content": "Page: C4A-Script API Reference\nSection: Multi-step Process\n\n```\n# Complex multi-step workflow\nPROC navigate_to_step\n  CLICK `.next-button`\n  WAIT `.step-content` 5\nENDPROC\n\n# Step 1\nWAIT `.step-1` 5\nSET `#name` \"John Doe\"\nnavigate_to_step\n\n# Step 2\nSET `#email`…\n```", "code_blocks": [{"language": "", "code": "# Complex multi-step workflow\nPROC navigate_to_step\n  CLICK `.next-button`\n  WAIT `.step-content` 5\nENDPROC\n\n# Step 1\nWAIT `.step-1` 5\nSET `#name` \"John Doe\"\nnavigate_to_step\n\n# Step 2\nSET `#email`…", "filename": ""}], "chunk_position": 23, "heading_path": "Multi-step Process > Multi-step Process", "breadcrumbs": "C4A-Script API Reference > Multi-step Process > Multi-step Process"}, {"id": "8fe68ed2b63901b5", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "Integration with Crawl4AI", "content": "Page: C4A-Script API Reference\nSection: Integration with Crawl4AI\n\nUse C4A-Script with Crawl4AI for dynamic content interaction:", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\n# Define interaction script\nscript = \"\"\"\n# Handle dynamic content loading\nWAIT `.content` 5\nIF (EXISTS `.load-more-button`) THEN CLICK…", "filename": ""}], "chunk_position": 23, "heading_path": "Integration with Crawl4AI > Integration with Crawl4AI", "breadcrumbs": "C4A-Script API Reference > Integration with Crawl4AI > Integration with Crawl4AI"}, {"id": "2984a1c37c9ebf19", "url": "https://docs.crawl4ai.com/api/c4a-script-reference/", "page_title": "C4A-Script API Reference", "page_type": "reference", "page_summary": "Complete reference for all C4A-Script commands, syntax, and advanced features, including navigation, waiting, mouse, keyboard, control flow, variables, procedures, and integration with Crawl4AI.", "heading": "Conclusion", "content": "Page: C4A-Script API Reference\nSection: Conclusion\n\nThis reference covers all available C4A-Script commands and patterns. For interactive learning, try the [tutorial](../examples/c4a_script/tutorial/) or [live…", "code_blocks": [], "chunk_position": 23, "heading_path": "Conclusion > Conclusion", "breadcrumbs": "C4A-Script API Reference > Conclusion > Conclusion"}, {"id": "8731bd6382adf1d5", "url": "https://docs.crawl4ai.com/api/crawl-result/", "page_title": "CrawlResult Reference", "page_type": "api", "page_summary": "Reference for the CrawlResult class in Crawl4AI, detailing all fields returned after a crawl operation, including content, metadata, links, media, and optional captures.", "heading": "Overview", "content": "Page: CrawlResult Reference\nSection: Overview\n\nThe `CrawlResult` class encapsulates everything returned after a single crawl operation. It provides the raw or processed content, details on links and media, plus optional metadata (like…", "code_blocks": [{"language": "python", "code": "class CrawlResult(BaseModel):\n    url: str\n    html: str\n    success: bool\n    cleaned_html: Optional[str] = None\n    fit_html: Optional[str] = None  # Preprocessed HTML optimized for extraction…", "filename": ""}], "chunk_position": 24, "heading_path": "Overview > Overview", "breadcrumbs": "CrawlResult Reference > Overview > Overview"}, {"id": "e4eb1b0365b76c36", "url": "https://docs.crawl4ai.com/api/crawl-result/", "page_title": "CrawlResult Reference", "page_type": "api", "page_summary": "Reference for the CrawlResult class in Crawl4AI, detailing all fields returned after a crawl operation, including content, metadata, links, media, and optional captures.", "heading": "1. Basic Crawl Info", "content": "Page: CrawlResult Reference\nSection: 1. Basic Crawl Info\n\n###### 1.1 `url` (str)\n\n**What**: The final crawled URL (after any redirects).\n\n###### 1.2 `success` (bool)\n\n**What**: `True` if the crawl pipeline ended without major errors; `False` otherwise.\n\n###### 1.3…", "code_blocks": [{"language": "python", "code": "print(result.url)  # e.g., \"https://example.com/\"", "filename": ""}, {"language": "python", "code": "if not result.success:\n    print(f\"Crawl failed: {result.error_message}\")", "filename": ""}, {"language": "python", "code": "if result.status_code == 404:\n    print(\"Page not found!\")", "filename": ""}, {"language": "python", "code": "if result.status_code in (301, 302) and result.redirected_status_code == 200:\n    print(f\"Redirected to {result.redirected_url} (OK)\")", "filename": ""}, {"language": "python", "code": "if not result.success:\n    print(\"Error:\", result.error_message)", "filename": ""}, {"language": "python", "code": "# If you used session_id=\"login_session\" in CrawlerRunConfig, see it here:\nprint(\"Session:\", result.session_id)", "filename": ""}, {"language": "python", "code": "if result.response_headers:\n    print(\"Server:\", result.response_headers.get(\"Server\", \"Unknown\"))", "filename": ""}, {"language": "python", "code": "if result.ssl_certificate:\n    print(\"Issuer:\", result.ssl_certificate.issuer)", "filename": ""}], "chunk_position": 24, "heading_path": "1. Basic Crawl Info > 1. Basic Crawl Info", "breadcrumbs": "CrawlResult Reference > 1. Basic Crawl Info > 1. Basic Crawl Info"}, {"id": "dc5c0cebc4f3d839", "url": "https://docs.crawl4ai.com/api/crawl-result/", "page_title": "CrawlResult Reference", "page_type": "api", "page_summary": "Reference for the CrawlResult class in Crawl4AI, detailing all fields returned after a crawl operation, including content, metadata, links, media, and optional captures.", "heading": "2. Raw / Cleaned Content", "content": "Page: CrawlResult Reference\nSection: 2. Raw / Cleaned Content\n\n###### 2.1 `html` (str)\n\n**What**: The **original** unmodified HTML from the final page load.\n\n###### 2.2 `cleaned_html` (Optional[str])\n\n**What**: A sanitized HTML version—scripts, styles, or excluded…", "code_blocks": [{"language": "python", "code": "# Possibly large\nprint(len(result.html))", "filename": ""}, {"language": "python", "code": "print(result.cleaned_html[:500])  # Show a snippet", "filename": ""}], "chunk_position": 24, "heading_path": "2. Raw / Cleaned Content > 2. Raw / Cleaned Content", "breadcrumbs": "CrawlResult Reference > 2. Raw / Cleaned Content > 2. Raw / Cleaned Content"}, {"id": "faa3d20b0c589f37", "url": "https://docs.crawl4ai.com/api/crawl-result/", "page_title": "CrawlResult Reference", "page_type": "api", "page_summary": "Reference for the CrawlResult class in Crawl4AI, detailing all fields returned after a crawl operation, including content, metadata, links, media, and optional captures.", "heading": "3. Markdown Fields", "content": "Page: CrawlResult Reference\nSection: 3. Markdown Fields\n\n###### 3.1 The Markdown Generation Approach\n\nCrawl4AI can convert HTML→Markdown, optionally including:\n\n- **Raw** markdown\n- **Links as citations** (with a references section)\n- **Fit** markdown if a…", "code_blocks": [{"language": "python", "code": "if result.markdown:\n    md_res = result.markdown\n    print(\"Raw MD:\", md_res.raw_markdown[:300])\n    print(\"Citations MD:\", md_res.markdown_with_citations[:300])\n    print(\"References:\",…", "filename": ""}, {"language": "python", "code": "print(result.markdown.raw_markdown[:200])\nprint(result.markdown.fit_markdown)\nprint(result.markdown.fit_html)", "filename": ""}], "chunk_position": 24, "heading_path": "3. Markdown Fields > 3. Markdown Fields", "breadcrumbs": "CrawlResult Reference > 3. Markdown Fields > 3. Markdown Fields"}, {"id": "ac930a183cf59c1f", "url": "https://docs.crawl4ai.com/api/crawl-result/", "page_title": "CrawlResult Reference", "page_type": "api", "page_summary": "Reference for the CrawlResult class in Crawl4AI, detailing all fields returned after a crawl operation, including content, metadata, links, media, and optional captures.", "heading": "4. Media & Links", "content": "Page: CrawlResult Reference\nSection: 4. Media & Links\n\n###### 4.1 `media` (Dict[str, List[Dict]])\n\n**What**: Contains info about discovered images, videos, or audio. Typically keys: `\"images\"`, `\"videos\"`, `\"audios\"`.\n\n**Common Fields** in each item:\n-…", "code_blocks": [{"language": "python", "code": "images = result.media.get(\"images\", [])\nfor img in images:\n    if img.get(\"score\", 0) > 5:\n        print(\"High-value image:\", img[\"src\"])", "filename": ""}, {"language": "python", "code": "for link in result.links[\"internal\"]:\n    print(f\"Internal link to {link['href']} with text {link['text']}\")", "filename": ""}], "chunk_position": 24, "heading_path": "4. Media & Links > 4. Media & Links", "breadcrumbs": "CrawlResult Reference > 4. Media & Links > 4. Media & Links"}, {"id": "bab00bd321ff3e5f", "url": "https://docs.crawl4ai.com/api/crawl-result/", "page_title": "CrawlResult Reference", "page_type": "api", "page_summary": "Reference for the CrawlResult class in Crawl4AI, detailing all fields returned after a crawl operation, including content, metadata, links, media, and optional captures.", "heading": "5. Additional Fields", "content": "Page: CrawlResult Reference\nSection: 5. Additional Fields\n\n###### 5.1 `extracted_content` (Optional[str])\n\n**What**: If you used `extraction_strategy` (CSS, LLM, etc.), the structured output (JSON).\n\n###### 5.2 `downloaded_files` (Optional[List[str]])\n\n**What**:…", "code_blocks": [{"language": "python", "code": "if result.extracted_content:\n    data = json.loads(result.extracted_content)\n    print(data)", "filename": ""}, {"language": "python", "code": "if result.downloaded_files:\n    for file_path in result.downloaded_files:\n        print(\"Downloaded:\", file_path)", "filename": ""}, {"language": "python", "code": "import base64\nif result.screenshot:\n    with open(\"page.png\", \"wb\") as f:\n        f.write(base64.b64decode(result.screenshot))", "filename": ""}, {"language": "python", "code": "if result.pdf:\n    with open(\"page.pdf\", \"wb\") as f:\n        f.write(result.pdf)", "filename": ""}, {"language": "python", "code": "if result.mhtml:\n    with open(\"page.mhtml\", \"w\", encoding=\"utf-8\") as f:\n        f.write(result.mhtml)", "filename": ""}, {"language": "python", "code": "if result.metadata:\n    print(\"Title:\", result.metadata.get(\"title\"))\n    print(\"Author:\", result.metadata.get(\"author\"))", "filename": ""}], "chunk_position": 24, "heading_path": "5. Additional Fields > 5. Additional Fields", "breadcrumbs": "CrawlResult Reference > 5. Additional Fields > 5. Additional Fields"}, {"id": "992842731c3025fd", "url": "https://docs.crawl4ai.com/api/crawl-result/", "page_title": "CrawlResult Reference", "page_type": "api", "page_summary": "Reference for the CrawlResult class in Crawl4AI, detailing all fields returned after a crawl operation, including content, metadata, links, media, and optional captures.", "heading": "6. `dispatch_result` (optional)", "content": "Page: CrawlResult Reference\nSection: 6. `dispatch_result` (optional)\n\nA `DispatchResult` object providing additional concurrency and resource usage information when crawling URLs in parallel (e.g., via `arun_many()` with custom dispatchers). It contains:\n\n- `task_id`:…", "code_blocks": [{"language": "python", "code": "# Example usage:\nfor result in results:\n    if result.success and result.dispatch_result:\n        dr = result.dispatch_result\n        print(f\"URL: {result.url}, Task ID: {dr.task_id}\")…", "filename": ""}], "chunk_position": 24, "heading_path": "6. `dispatch_result` (optional) > 6. `dispatch_result` (optional)", "breadcrumbs": "CrawlResult Reference > 6. `dispatch_result` (optional) > 6. `dispatch_result` (optional)"}, {"id": "cedd87b2a617cde7", "url": "https://docs.crawl4ai.com/api/crawl-result/", "page_title": "CrawlResult Reference", "page_type": "api", "page_summary": "Reference for the CrawlResult class in Crawl4AI, detailing all fields returned after a crawl operation, including content, metadata, links, media, and optional captures.", "heading": "7. Network Requests & Console Messages", "content": "Page: CrawlResult Reference\nSection: 7. Network Requests & Console Messages\n\nWhen you enable network and console message capturing in `CrawlerRunConfig` using `capture_network_requests=True` and `capture_console_messages=True`, the `CrawlResult` will include these…", "code_blocks": [{"language": "python", "code": "if result.network_requests:\n    # Count different types of events\n    requests = [r for r in result.network_requests if r.get(\"event_type\") == \"request\"]\n    responses = [r for r in…", "filename": ""}, {"language": "python", "code": "if result.console_messages:\n    # Count messages by type\n    message_types = {}\n    for msg in result.console_messages:\n        msg_type = msg.get(\"type\", \"unknown\")\n        message_types[msg_type] =…", "filename": ""}], "chunk_position": 24, "heading_path": "7. Network Requests & Console Messages > 7. Network Requests & Console Messages", "breadcrumbs": "CrawlResult Reference > 7. Network Requests & Console Messages > 7. Network Requests & Console Messages"}, {"id": "a56498ea06bb06a4", "url": "https://docs.crawl4ai.com/api/crawl-result/", "page_title": "CrawlResult Reference", "page_type": "api", "page_summary": "Reference for the CrawlResult class in Crawl4AI, detailing all fields returned after a crawl operation, including content, metadata, links, media, and optional captures.", "heading": "8. Example: Accessing Everything", "content": "Page: CrawlResult Reference\nSection: 8. Example: Accessing Everything\n\n```python\nasync def handle_result(result: CrawlResult):\n    if not result.success:\n        print(\"Crawl error:\", result.error_message)\n        return\n\n    # Basic info\n    print(\"Crawled URL:\", result.url)…\n```", "code_blocks": [{"language": "python", "code": "async def handle_result(result: CrawlResult):\n    if not result.success:\n        print(\"Crawl error:\", result.error_message)\n        return\n\n    # Basic info\n    print(\"Crawled URL:\", result.url)…", "filename": ""}], "chunk_position": 24, "heading_path": "8. Example: Accessing Everything > 8. Example: Accessing Everything", "breadcrumbs": "CrawlResult Reference > 8. Example: Accessing Everything > 8. Example: Accessing Everything"}, {"id": "ac1543e706961127", "url": "https://docs.crawl4ai.com/api/crawl-result/", "page_title": "CrawlResult Reference", "page_type": "api", "page_summary": "Reference for the CrawlResult class in Crawl4AI, detailing all fields returned after a crawl operation, including content, metadata, links, media, and optional captures.", "heading": "9. Key Points & Future", "content": "Page: CrawlResult Reference\nSection: 9. Key Points & Future\n\n1. **Deprecated legacy properties of CrawlResult**\n   - `markdown_v2` - Removed in v0.5 and now raises `AttributeError`. Use `result.markdown` instead.\n   - `fit_markdown` and `fit_html` - No longer…", "code_blocks": [], "chunk_position": 24, "heading_path": "9. Key Points & Future > 9. Key Points & Future", "breadcrumbs": "CrawlResult Reference > 9. Key Points & Future > 9. Key Points & Future"}, {"id": "70dd1343446caf1d", "url": "https://docs.crawl4ai.com/api/digest/", "page_title": "digest()", "page_type": "api", "page_summary": "The digest() method is the primary interface for adaptive web crawling. It intelligently crawls websites starting from a given URL, guided by a query, and automatically determines when sufficient…", "heading": "Method Signature", "content": "Page: digest()\nSection: Method Signature\n\n```python\nasync def digest(\n    start_url: str,\n    query: str,\n    resume_from: Optional[Union[str, Path]] = None\n) -> CrawlState\n```", "code_blocks": [{"language": "python", "code": "async def digest(\n    start_url: str,\n    query: str,\n    resume_from: Optional[Union[str, Path]] = None\n) -> CrawlState", "filename": ""}], "chunk_position": 25, "heading_path": "Method Signature > Method Signature", "breadcrumbs": "digest() > Method Signature > Method Signature"}, {"id": "a6aac8012480b93e", "url": "https://docs.crawl4ai.com/api/digest/", "page_title": "digest()", "page_type": "api", "page_summary": "The digest() method is the primary interface for adaptive web crawling. It intelligently crawls websites starting from a given URL, guided by a query, and automatically determines when sufficient…", "heading": "Parameters", "content": "Page: digest()\nSection: Parameters\n\n###### start_url\n\n- **Type** : `str`\n- **Required** : Yes\n- **Description** : The starting URL for the crawl. This should be a valid HTTP/HTTPS URL that serves as the entry point for information…", "code_blocks": [], "chunk_position": 25, "heading_path": "Parameters > Parameters", "breadcrumbs": "digest() > Parameters > Parameters"}, {"id": "967bcc230862b127", "url": "https://docs.crawl4ai.com/api/digest/", "page_title": "digest()", "page_type": "api", "page_summary": "The digest() method is the primary interface for adaptive web crawling. It intelligently crawls websites starting from a given URL, guided by a query, and automatically determines when sufficient…", "heading": "Return Value", "content": "Page: digest()\nSection: Return Value\n\nReturns a `CrawlState` object containing:\n\n- **crawled_urls** (`Set[str]`): All URLs that have been crawled\n- **knowledge_base** (`List[CrawlResult]`): Collection of crawled pages with content\n-…", "code_blocks": [], "chunk_position": 25, "heading_path": "Return Value > Return Value", "breadcrumbs": "digest() > Return Value > Return Value"}, {"id": "fb6c2000172ff550", "url": "https://docs.crawl4ai.com/api/digest/", "page_title": "digest()", "page_type": "api", "page_summary": "The digest() method is the primary interface for adaptive web crawling. It intelligently crawls websites starting from a given URL, guided by a query, and automatically determines when sufficient…", "heading": "How It Works", "content": "Page: digest()\nSection: How It Works\n\nThe `digest()` method implements an intelligent crawling algorithm:\n\n- **Initial Crawl** : Starts from the provided URL\n- **Link Analysis** : Evaluates all discovered links for relevance\n-…", "code_blocks": [], "chunk_position": 25, "heading_path": "How It Works > How It Works", "breadcrumbs": "digest() > How It Works > How It Works"}, {"id": "2e44960755ed5f5c", "url": "https://docs.crawl4ai.com/api/digest/", "page_title": "digest()", "page_type": "api", "page_summary": "The digest() method is the primary interface for adaptive web crawling. It intelligently crawls websites starting from a given URL, guided by a query, and automatically determines when sufficient…", "heading": "Examples", "content": "Page: digest()\nSection: Examples\n\n```python\nasync with AsyncWebCrawler() as crawler:\n    adaptive = AdaptiveCrawler(crawler)\n\n    state = await adaptive.digest(\n        start_url=\"https://docs.python.org/3/\",\n        query=\"async await context…\n```\n\n```python\nconfig = AdaptiveConfig(\n    confidence_threshold=0.9,  # Require high confidence\n    max_pages=30,             # Allow more pages\n    top_k_links=3             # Follow top 3 links per…\n```\n\n```python\n# First crawl - may be interrupted\nstate1 = await adaptive.digest(\n    start_url=\"https://example.com\",\n    query=\"machine learning algorithms\"\n)\n\n# Save state (if not…\n```\n\n```python\nstate = await adaptive.digest(\n    start_url=\"https://docs.example.com\",\n    query=\"api reference\"\n)\n\n# Monitor progress\nprint(f\"Pages crawled: {len(state.crawled_urls)}\")\nprint(f\"New terms…\n```", "code_blocks": [{"language": "python", "code": "async with AsyncWebCrawler() as crawler:\n    adaptive = AdaptiveCrawler(crawler)\n\n    state = await adaptive.digest(\n        start_url=\"https://docs.python.org/3/\",\n        query=\"async await context…", "filename": ""}, {"language": "python", "code": "config = AdaptiveConfig(\n    confidence_threshold=0.9,  # Require high confidence\n    max_pages=30,             # Allow more pages\n    top_k_links=3             # Follow top 3 links per…", "filename": ""}, {"language": "python", "code": "# First crawl - may be interrupted\nstate1 = await adaptive.digest(\n    start_url=\"https://example.com\",\n    query=\"machine learning algorithms\"\n)\n\n# Save state (if not…", "filename": ""}, {"language": "python", "code": "state = await adaptive.digest(\n    start_url=\"https://docs.example.com\",\n    query=\"api reference\"\n)\n\n# Monitor progress\nprint(f\"Pages crawled: {len(state.crawled_urls)}\")\nprint(f\"New terms…", "filename": ""}], "chunk_position": 25, "heading_path": "Examples > Examples", "breadcrumbs": "digest() > Examples > Examples"}, {"id": "b7d9627ec39f59bb", "url": "https://docs.crawl4ai.com/api/digest/", "page_title": "digest()", "page_type": "api", "page_summary": "The digest() method is the primary interface for adaptive web crawling. It intelligently crawls websites starting from a given URL, guided by a query, and automatically determines when sufficient…", "heading": "Query Best Practices", "content": "Page: digest()\nSection: Query Best Practices\n\n- **Be Specific** : Use descriptive terms that appear in target content\n- **Include Key Terms** : Add technical terms you expect to find\n- **Multiple Concepts** : Combine related concepts for…", "code_blocks": [{"language": "python", "code": "# Good\nquery = \"python async context managers implementation\"\n\n# Too broad\nquery = \"python programming\"", "filename": ""}, {"language": "python", "code": "query = \"oauth2 jwt refresh tokens authorization\"", "filename": ""}, {"language": "python", "code": "query = \"rest api pagination sorting filtering\"", "filename": ""}], "chunk_position": 25, "heading_path": "Query Best Practices > Query Best Practices", "breadcrumbs": "digest() > Query Best Practices > Query Best Practices"}, {"id": "edb260f9d5a015c8", "url": "https://docs.crawl4ai.com/api/digest/", "page_title": "digest()", "page_type": "api", "page_summary": "The digest() method is the primary interface for adaptive web crawling. It intelligently crawls websites starting from a given URL, guided by a query, and automatically determines when sufficient…", "heading": "Performance Considerations", "content": "Page: digest()\nSection: Performance Considerations\n\n- **Initial URL** : Choose a page with good navigation (e.g., documentation index)\n- **Query Length** : 3-8 terms typically work best\n- **Link Density** : Sites with clear navigation crawl more…", "code_blocks": [], "chunk_position": 25, "heading_path": "Performance Considerations > Performance Considerations", "breadcrumbs": "digest() > Performance Considerations > Performance Considerations"}, {"id": "ed176e0d2c5bb0af", "url": "https://docs.crawl4ai.com/api/digest/", "page_title": "digest()", "page_type": "api", "page_summary": "The digest() method is the primary interface for adaptive web crawling. It intelligently crawls websites starting from a given URL, guided by a query, and automatically determines when sufficient…", "heading": "Error Handling", "content": "Page: digest()\nSection: Error Handling\n\n```python\ntry:\n    state = await adaptive.digest(\n        start_url=\"https://example.com\",\n        query=\"search terms\"\n    )\nexcept Exception as e:\n    print(f\"Crawl failed: {e}\")\n    # State is auto-saved if…\n```", "code_blocks": [{"language": "python", "code": "try:\n    state = await adaptive.digest(\n        start_url=\"https://example.com\",\n        query=\"search terms\"\n    )\nexcept Exception as e:\n    print(f\"Crawl failed: {e}\")\n    # State is auto-saved if…", "filename": ""}], "chunk_position": 25, "heading_path": "Error Handling > Error Handling", "breadcrumbs": "digest() > Error Handling > Error Handling"}, {"id": "b848d676990e7718", "url": "https://docs.crawl4ai.com/api/digest/", "page_title": "digest()", "page_type": "api", "page_summary": "The digest() method is the primary interface for adaptive web crawling. It intelligently crawls websites starting from a given URL, guided by a query, and automatically determines when sufficient…", "heading": "Stopping Conditions", "content": "Page: digest()\nSection: Stopping Conditions\n\nThe crawl stops when any of these conditions are met:\n\n- **Confidence Threshold** : Reached the configured confidence level\n- **Page Limit** : Crawled the maximum number of pages\n- **Diminishing…", "code_blocks": [], "chunk_position": 25, "heading_path": "Stopping Conditions > Stopping Conditions", "breadcrumbs": "digest() > Stopping Conditions > Stopping Conditions"}, {"id": "7243d807934c2e0a", "url": "https://docs.crawl4ai.com/api/digest/", "page_title": "digest()", "page_type": "api", "page_summary": "The digest() method is the primary interface for adaptive web crawling. It intelligently crawls websites starting from a given URL, guided by a query, and automatically determines when sufficient…", "heading": "See Also", "content": "Page: digest()\nSection: See Also\n\n- [AdaptiveCrawler Class](../adaptive-crawler/)\n- [Adaptive Crawling Guide](../../core/adaptive-crawling/)\n- [Configuration Options](../../core/adaptive-crawling/#configuration-options)", "code_blocks": [], "chunk_position": 25, "heading_path": "See Also > See Also", "breadcrumbs": "digest() > See Also > See Also"}, {"id": "be9d96c7e0521c4f", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "1. BrowserConfig – Controlling the Browser", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: 1. BrowserConfig – Controlling the Browser\n\n`BrowserConfig` focuses on  **how**  the browser is launched and behaves. This includes headless mode, proxies, user agents, and other environment tweaks.", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler, BrowserConfig\n\nbrowser_cfg = BrowserConfig(\n    browser_type=\"chromium\",\n    headless=True,\n    viewport_width=1280,\n    viewport_height=720,…", "filename": ""}], "chunk_position": 26, "heading_path": "1. BrowserConfig – Controlling the Browser > 1. BrowserConfig – Controlling the Browser", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > 1. BrowserConfig – Controlling the Browser > 1. BrowserConfig – Controlling the Browser"}, {"id": "fb19c8d358db903b", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "1.1 Parameter Highlights", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: 1.1 Parameter Highlights\n\n| **Parameter** | **Type / Default** | **What It Does** |\n| --- | --- | --- |\n| **`browser_type`** | `\"chromium\"`, `\"firefox\"`, `\"webkit\"`  *(default: `\"chromium\"`)* | Which browser engine to use.…", "code_blocks": [], "chunk_position": 26, "heading_path": "1.1 Parameter Highlights > 1.1 Parameter Highlights", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > 1.1 Parameter Highlights > 1.1 Parameter Highlights"}, {"id": "af73a52a6273aee9", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "2. CrawlerRunConfig – Controlling Each Crawl", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: 2. CrawlerRunConfig – Controlling Each Crawl\n\nWhile `BrowserConfig` sets up the  **environment** , `CrawlerRunConfig` details  **how**  each  **crawl operation**  should behave: caching, content filtering, link or domain blocking, timeouts,…", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nrun_cfg = CrawlerRunConfig(\n    wait_for=\"css:.main-content\",\n    word_count_threshold=15,\n    excluded_tags=[\"nav\", \"footer\"],…", "filename": ""}], "chunk_position": 26, "heading_path": "2. CrawlerRunConfig – Controlling Each Crawl > 2. CrawlerRunConfig – Controlling Each Crawl", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > 2. CrawlerRunConfig – Controlling Each Crawl > 2. CrawlerRunConfig – Controlling Each Crawl"}, {"id": "ae4fc4649f6f2b55", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "A) Content Processing", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: A) Content Processing\n\n| **Parameter** | **Type / Default** | **What It Does** |\n| --- | --- | --- |\n| **`word_count_threshold`** | `int` (default: ~200) | Skips text blocks below X words. Helps ignore trivial sections.…", "code_blocks": [], "chunk_position": 26, "heading_path": "A) Content Processing > A) Content Processing", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > A) Content Processing > A) Content Processing"}, {"id": "3a22a73ef70ff471", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "B) Browser Location and Identity", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: B) Browser Location and Identity\n\n| **Parameter** | **Type / Default** | **What It Does** |\n| --- | --- | --- |\n| **`locale`** | `str or None` (None) | Browser's locale (e.g., \"en-US\", \"fr-FR\") for language preferences. |\n|…", "code_blocks": [], "chunk_position": 26, "heading_path": "B) Browser Location and Identity > B) Browser Location and Identity", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > B) Browser Location and Identity > B) Browser Location and Identity"}, {"id": "8089a335f1ad392d", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "C) Caching & Session", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: C) Caching & Session\n\n| **Parameter** | **Type / Default** | **What It Does** |\n| --- | --- | --- |\n| **`cache_mode`** | `CacheMode or None` | Controls how caching is handled (`ENABLED`, `BYPASS`, `DISABLED`, etc.). If…", "code_blocks": [], "chunk_position": 26, "heading_path": "C) Caching & Session > C) Caching & Session", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > C) Caching & Session > C) Caching & Session"}, {"id": "71c8a315cb824c0f", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "D) Page Navigation & Timing", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: D) Page Navigation & Timing\n\n| **Parameter** | **Type / Default** | **What It Does** |\n| --- | --- | --- |\n| **`wait_until`** | `str` (domcontentloaded) | Condition for navigation to \"complete\". Often `\"networkidle\"` or…", "code_blocks": [], "chunk_position": 26, "heading_path": "D) Page Navigation & Timing > D) Page Navigation & Timing", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > D) Page Navigation & Timing > D) Page Navigation & Timing"}, {"id": "1849b9252e89ebfa", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "E) Page Interaction", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: E) Page Interaction\n\n| **Parameter** | **Type / Default** | **What It Does** |\n| --- | --- | --- |\n| **`js_code`** | `str or list[str]` (None) | JavaScript to run  **after**  `wait_for` and `delay_before_return_html`, on…", "code_blocks": [], "chunk_position": 26, "heading_path": "E) Page Interaction > E) Page Interaction", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > E) Page Interaction > E) Page Interaction"}, {"id": "af9b128e22e1b9e7", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "F) Media Handling", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: F) Media Handling\n\n| **Parameter** | **Type / Default** | **What It Does** |\n| --- | --- | --- |\n| **`screenshot`** | `bool` (False) | Capture a screenshot (base64) in `result.screenshot`. |\n| **`screenshot_wait_for`**…", "code_blocks": [], "chunk_position": 26, "heading_path": "F) Media Handling > F) Media Handling", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > F) Media Handling > F) Media Handling"}, {"id": "d06e36cad2642f99", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "G) Link/Domain Handling", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: G) Link/Domain Handling\n\n| **Parameter** | **Type / Default** | **What It Does** |\n| --- | --- | --- |\n| **`exclude_social_media_domains`** | `list` (e.g. Facebook/Twitter) | A default list can be extended. Any link to these…", "code_blocks": [], "chunk_position": 26, "heading_path": "G) Link/Domain Handling > G) Link/Domain Handling", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > G) Link/Domain Handling > G) Link/Domain Handling"}, {"id": "bc558f86e083ec75", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "H) Debug, Logging & Network Monitoring", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: H) Debug, Logging & Network Monitoring\n\n| **Parameter** | **Type / Default** | **What It Does** |\n| --- | --- | --- |\n| **`verbose`** | `bool` (True) | Prints logs detailing each step of crawling, interactions, or errors. |\n|…", "code_blocks": [], "chunk_position": 26, "heading_path": "H) Debug, Logging & Network Monitoring > H) Debug, Logging & Network Monitoring", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > H) Debug, Logging & Network Monitoring > H) Debug, Logging & Network Monitoring"}, {"id": "b70906d5ff146554", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "I) Connection & HTTP Parameters", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: I) Connection & HTTP Parameters\n\n| **Parameter** | **Type / Default** | **What It Does** |\n| --- | --- | --- |\n| **`method`** | `str` (\"GET\") | HTTP method to use when using AsyncHTTPCrawlerStrategy (e.g., \"GET\", \"POST\"). |\n|…", "code_blocks": [], "chunk_position": 26, "heading_path": "I) Connection & HTTP Parameters > I) Connection & HTTP Parameters", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > I) Connection & HTTP Parameters > I) Connection & HTTP Parameters"}, {"id": "73507a117675bca9", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "J) Virtual Scroll Configuration", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: J) Virtual Scroll Configuration\n\n| **Parameter** | **Type / Default** | **What It Does** |\n| --- | --- | --- |\n| **`virtual_scroll_config`** | `VirtualScrollConfig or dict` (None) | Configuration for handling virtualized scrolling…", "code_blocks": [{"language": "python", "code": "from crawl4ai import VirtualScrollConfig\n\nvirtual_config = VirtualScrollConfig(\n    container_selector=\"#timeline\",    # CSS selector for scrollable container\n    scroll_count=30,                   #…", "filename": ""}], "chunk_position": 26, "heading_path": "J) Virtual Scroll Configuration > J) Virtual Scroll Configuration", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > J) Virtual Scroll Configuration > J) Virtual Scroll Configuration"}, {"id": "e1fa1a7d98e1fc5d", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "K) URL Matching Configuration", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: K) URL Matching Configuration\n\n| **Parameter** | **Type / Default** | **What It Does** |\n| --- | --- | --- |\n| **`url_matcher`** | `UrlMatcher` (None) | Pattern(s) to match URLs against. Can be: string (glob), function, or list of…", "code_blocks": [{"language": "python", "code": "from crawl4ai import CrawlerRunConfig, MatchMode\nfrom crawl4ai.processors.pdf import PDFContentScrapingStrategy\nfrom crawl4ai.extraction_strategy import JsonCssExtractionStrategy\n\n# Simple string…", "filename": ""}], "chunk_position": 26, "heading_path": "K) URL Matching Configuration > K) URL Matching Configuration", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > K) URL Matching Configuration > K) URL Matching Configuration"}, {"id": "966fe48404b4e86f", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "L) Advanced Crawling Features", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: L) Advanced Crawling Features\n\n| **Parameter** | **Type / Default** | **What It Does** |\n| --- | --- | --- |\n| **`deep_crawl_strategy`** | `DeepCrawlStrategy or None` (None) | Strategy for deep/recursive crawling. Enables…", "code_blocks": [], "chunk_position": 26, "heading_path": "L) Advanced Crawling Features > L) Advanced Crawling Features", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > L) Advanced Crawling Features > L) Advanced Crawling Features"}, {"id": "4a4ad16386b455bb", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "2.2 Helper Methods", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: 2.2 Helper Methods\n\nBoth `BrowserConfig` and `CrawlerRunConfig` provide a `clone()` method to create modified copies:", "code_blocks": [{"language": "python", "code": "# Create a base configuration\nbase_config = CrawlerRunConfig(\n    cache_mode=CacheMode.ENABLED,\n    word_count_threshold=200\n)\n\n# Create variations using clone()\nstream_config =…", "filename": ""}], "chunk_position": 26, "heading_path": "2.2 Helper Methods > 2.2 Helper Methods", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > 2.2 Helper Methods > 2.2 Helper Methods"}, {"id": "bcb5524adbe1ebbf", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "Class-Level Defaults (`set_defaults` / `get_defaults` / `reset_defaults`)", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: Class-Level Defaults (`set_defaults` / `get_defaults` / `reset_defaults`)\n\nBoth config classes support class-level default overrides. When deploying in a server or cloud context, this eliminates the need to pass the same parameters at every call site.\n\n **Resolution…", "code_blocks": [{"language": "python", "code": "from crawl4ai import BrowserConfig, CrawlerRunConfig\n\n# Set once at application startup\nBrowserConfig.set_defaults(\n    cache_cdp_connection=True,\n    cdp_close_delay=0,…", "filename": ""}], "chunk_position": 26, "heading_path": "Class-Level Defaults (`set_defaults` / `get_defaults` / `reset_defaults`) > Class-Level Defaults (`set_defaults` / `get_defaults` / `reset_defaults`)", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > Class-Level Defaults (`set_defaults` / `get_defaults` / `reset_defaults`) > Class-Level Defaults (`set_defaults` / `get_defaults` / `reset_defaults`)"}, {"id": "b1c49140a929e87c", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "2.3 Example Usage", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: 2.3 Example Usage\n\n```python\nimport asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode\n\nasync def main():\n    # Configure the browser\n    browser_cfg = BrowserConfig(…\n```", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode\n\nasync def main():\n    # Configure the browser\n    browser_cfg = BrowserConfig(…", "filename": ""}], "chunk_position": 26, "heading_path": "2.3 Example Usage > 2.3 Example Usage", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > 2.3 Example Usage > 2.3 Example Usage"}, {"id": "06a033fe8d67ffa1", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "2.4 Compliance & Ethics", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: 2.4 Compliance & Ethics\n\n| **Parameter** | **Type / Default** | **What It Does** |\n| --- | --- | --- |\n| **`check_robots_txt`** | `bool` (False) | When True, checks and respects robots.txt rules before crawling. Uses…", "code_blocks": [{"language": "python", "code": "run_config = CrawlerRunConfig(\n    check_robots_txt=True,  # Enable robots.txt compliance\n    user_agent=\"MyBot/1.0\"  # Identify your crawler\n)", "filename": ""}], "chunk_position": 26, "heading_path": "2.4 Compliance & Ethics > 2.4 Compliance & Ethics", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > 2.4 Compliance & Ethics > 2.4 Compliance & Ethics"}, {"id": "d94222bc7179cc39", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "3. LLMConfig - Setting up LLM providers", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: 3. LLMConfig - Setting up LLM providers\n\nLLMConfig is useful to pass LLM provider config to strategies and functions that rely on LLMs to do extraction, filtering, schema generation etc. Currently it can be used in the following -\n\n-…", "code_blocks": [], "chunk_position": 26, "heading_path": "3. LLMConfig - Setting up LLM providers > 3. LLMConfig - Setting up LLM providers", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > 3. LLMConfig - Setting up LLM providers > 3. LLMConfig - Setting up LLM providers"}, {"id": "5e024dd763be9bd9", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "3.1 Parameters", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: 3.1 Parameters\n\n| **Parameter** | **Type / Default** | **What It Does** |\n| --- | --- | --- |\n| **`provider`** | `\"ollama/llama3\",\"groq/llama3-70b-8192\",\"groq/llama3-8b-8192\", \"openai/gpt-4o-mini\"…", "code_blocks": [], "chunk_position": 26, "heading_path": "3.1 Parameters > 3.1 Parameters", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > 3.1 Parameters > 3.1 Parameters"}, {"id": "47df31a1b427a3eb", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "3.2 Example Usage", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: 3.2 Example Usage\n\n```python\nllm_config = LLMConfig(\n    provider=\"openai/gpt-4o-mini\",\n    api_token=os.getenv(\"OPENAI_API_KEY\"),\n    backoff_base_delay=1, # optional\n    backoff_max_attempts=5, # optional…\n```", "code_blocks": [{"language": "python", "code": "llm_config = LLMConfig(\n    provider=\"openai/gpt-4o-mini\",\n    api_token=os.getenv(\"OPENAI_API_KEY\"),\n    backoff_base_delay=1, # optional\n    backoff_max_attempts=5, # optional…", "filename": ""}], "chunk_position": 26, "heading_path": "3.2 Example Usage > 3.2 Example Usage", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > 3.2 Example Usage > 3.2 Example Usage"}, {"id": "ad2a5307586f27c0", "url": "https://docs.crawl4ai.com/api/parameters/", "page_title": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the configuration classes for Crawl4AI: BrowserConfig, CrawlerRunConfig, and LLMConfig, detailing their parameters, usage, and helper methods.", "heading": "4. Putting It All Together", "content": "Page: Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x)\nSection: 4. Putting It All Together\n\n- **Use**  `BrowserConfig` for  **global**  browser settings: engine, headless, proxy, user agent.\n- **Use**  `CrawlerRunConfig` for each crawl’s  **context** : how to filter content, handle caching,…", "code_blocks": [{"language": "python", "code": "# Create a modified copy with the clone() method\nstream_cfg = run_cfg.clone(\n    stream=True,\n    cache_mode=CacheMode.BYPASS\n)\n\n# Or set project-wide defaults once at…", "filename": ""}], "chunk_position": 26, "heading_path": "4. Putting It All Together > 4. Putting It All Together", "breadcrumbs": "Browser, Crawler & LLM Config - Crawl4AI Documentation (v0.9.x) > 4. Putting It All Together > 4. Putting It All Together"}, {"id": "e7a893e40fef9299", "url": "https://docs.crawl4ai.com/api/strategies/", "page_title": "Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "API reference for Crawl4AI extraction and chunking strategies, covering LLMExtractionStrategy, RegexExtractionStrategy, CosineStrategy, JsonCssExtractionStrategy, and various chunking strategies with…", "heading": "Extraction Strategies", "content": "Page: Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Extraction Strategies\n\nAll extraction strategies inherit from the base `ExtractionStrategy` class and implement two key methods:\n- `extract(url: str, html: str) -> List[Dict[str, Any]]`\n- `run(url: str, sections:…", "code_blocks": [], "chunk_position": 27, "heading_path": "Extraction Strategies > Extraction Strategies", "breadcrumbs": "Strategies - Crawl4AI Documentation (v0.9.x) > Extraction Strategies > Extraction Strategies"}, {"id": "25f4fed609fb7fab", "url": "https://docs.crawl4ai.com/api/strategies/", "page_title": "Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "API reference for Crawl4AI extraction and chunking strategies, covering LLMExtractionStrategy, RegexExtractionStrategy, CosineStrategy, JsonCssExtractionStrategy, and various chunking strategies with…", "heading": "LLMExtractionStrategy", "content": "Page: Strategies - Crawl4AI Documentation (v0.9.x)\nSection: LLMExtractionStrategy\n\nUsed for extracting structured data using Language Models.", "code_blocks": [{"language": "python", "code": "LLMExtractionStrategy(\n    # Required Parameters\n    provider: str = DEFAULT_PROVIDER,     # LLM provider (e.g., \"ollama/llama2\")\n    api_token: Optional[str] = None,      # API token\n\n    #…", "filename": ""}], "chunk_position": 27, "heading_path": "LLMExtractionStrategy > LLMExtractionStrategy", "breadcrumbs": "Strategies - Crawl4AI Documentation (v0.9.x) > LLMExtractionStrategy > LLMExtractionStrategy"}, {"id": "cf19383d637ad3b2", "url": "https://docs.crawl4ai.com/api/strategies/", "page_title": "Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "API reference for Crawl4AI extraction and chunking strategies, covering LLMExtractionStrategy, RegexExtractionStrategy, CosineStrategy, JsonCssExtractionStrategy, and various chunking strategies with…", "heading": "RegexExtractionStrategy", "content": "Page: Strategies - Crawl4AI Documentation (v0.9.x)\nSection: RegexExtractionStrategy\n\nUsed for fast pattern-based extraction of common entities using regular expressions.", "code_blocks": [{"language": "python", "code": "RegexExtractionStrategy(\n    # Pattern Configuration\n    pattern: IntFlag = RegexExtractionStrategy.Nothing,  # Bit flags of built-in patterns to use\n    custom: Optional[Dict[str, str]] = None,…", "filename": ""}], "chunk_position": 27, "heading_path": "RegexExtractionStrategy > RegexExtractionStrategy", "breadcrumbs": "Strategies - Crawl4AI Documentation (v0.9.x) > RegexExtractionStrategy > RegexExtractionStrategy"}, {"id": "eef5c576e7ea09e6", "url": "https://docs.crawl4ai.com/api/strategies/", "page_title": "Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "API reference for Crawl4AI extraction and chunking strategies, covering LLMExtractionStrategy, RegexExtractionStrategy, CosineStrategy, JsonCssExtractionStrategy, and various chunking strategies with…", "heading": "CosineStrategy", "content": "Page: Strategies - Crawl4AI Documentation (v0.9.x)\nSection: CosineStrategy\n\nUsed for content similarity-based extraction and clustering.", "code_blocks": [{"language": "python", "code": "CosineStrategy(\n    # Content Filtering\n    semantic_filter: str = None,        # Topic/keyword filter\n    word_count_threshold: int = 10,     # Minimum words per cluster\n    sim_threshold: float =…", "filename": ""}], "chunk_position": 27, "heading_path": "CosineStrategy > CosineStrategy", "breadcrumbs": "Strategies - Crawl4AI Documentation (v0.9.x) > CosineStrategy > CosineStrategy"}, {"id": "172f53bf1acb2d25", "url": "https://docs.crawl4ai.com/api/strategies/", "page_title": "Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "API reference for Crawl4AI extraction and chunking strategies, covering LLMExtractionStrategy, RegexExtractionStrategy, CosineStrategy, JsonCssExtractionStrategy, and various chunking strategies with…", "heading": "JsonCssExtractionStrategy", "content": "Page: Strategies - Crawl4AI Documentation (v0.9.x)\nSection: JsonCssExtractionStrategy\n\nUsed for CSS selector-based structured data extraction.", "code_blocks": [{"language": "python", "code": "JsonCssExtractionStrategy(\n    schema: Dict[str, Any],    # Extraction schema\n    verbose: bool = False      # Enable verbose logging\n)\n\n# Schema Structure\nschema = {\n    \"name\": str,              #…", "filename": ""}], "chunk_position": 27, "heading_path": "JsonCssExtractionStrategy > JsonCssExtractionStrategy", "breadcrumbs": "Strategies - Crawl4AI Documentation (v0.9.x) > JsonCssExtractionStrategy > JsonCssExtractionStrategy"}, {"id": "31df81b42ee5a6dc", "url": "https://docs.crawl4ai.com/api/strategies/", "page_title": "Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "API reference for Crawl4AI extraction and chunking strategies, covering LLMExtractionStrategy, RegexExtractionStrategy, CosineStrategy, JsonCssExtractionStrategy, and various chunking strategies with…", "heading": "Chunking Strategies", "content": "Page: Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Chunking Strategies\n\nAll chunking strategies inherit from `ChunkingStrategy` and implement the `chunk(text: str) -> list` method.", "code_blocks": [], "chunk_position": 27, "heading_path": "Chunking Strategies > Chunking Strategies", "breadcrumbs": "Strategies - Crawl4AI Documentation (v0.9.x) > Chunking Strategies > Chunking Strategies"}, {"id": "b4d13c53d3900719", "url": "https://docs.crawl4ai.com/api/strategies/", "page_title": "Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "API reference for Crawl4AI extraction and chunking strategies, covering LLMExtractionStrategy, RegexExtractionStrategy, CosineStrategy, JsonCssExtractionStrategy, and various chunking strategies with…", "heading": "SlidingWindowChunking", "content": "Page: Strategies - Crawl4AI Documentation (v0.9.x)\nSection: SlidingWindowChunking\n\nCreates overlapping chunks with a sliding window approach.", "code_blocks": [{"language": "python", "code": "SlidingWindowChunking(\n    window_size: int = 100,    # Window size in words\n    step: int = 50             # Step size between windows\n)", "filename": ""}], "chunk_position": 27, "heading_path": "SlidingWindowChunking > SlidingWindowChunking", "breadcrumbs": "Strategies - Crawl4AI Documentation (v0.9.x) > SlidingWindowChunking > SlidingWindowChunking"}, {"id": "4eab3a118824421b", "url": "https://docs.crawl4ai.com/api/strategies/", "page_title": "Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "API reference for Crawl4AI extraction and chunking strategies, covering LLMExtractionStrategy, RegexExtractionStrategy, CosineStrategy, JsonCssExtractionStrategy, and various chunking strategies with…", "heading": "LLM Extraction", "content": "Page: Strategies - Crawl4AI Documentation (v0.9.x)\nSection: LLM Extraction\n\n```python\nfrom pydantic import BaseModel\nfrom crawl4ai import LLMExtractionStrategy\nfrom crawl4ai import LLMConfig\n\n# Define schema\nclass Article(BaseModel):\n    title: str\n    content: str\n    author: str\n\n#…\n```", "code_blocks": [{"language": "python", "code": "from pydantic import BaseModel\nfrom crawl4ai import LLMExtractionStrategy\nfrom crawl4ai import LLMConfig\n\n# Define schema\nclass Article(BaseModel):\n    title: str\n    content: str\n    author: str\n\n#…", "filename": ""}], "chunk_position": 27, "heading_path": "LLM Extraction > LLM Extraction", "breadcrumbs": "Strategies - Crawl4AI Documentation (v0.9.x) > LLM Extraction > LLM Extraction"}, {"id": "66101c0af76f295c", "url": "https://docs.crawl4ai.com/api/strategies/", "page_title": "Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "API reference for Crawl4AI extraction and chunking strategies, covering LLMExtractionStrategy, RegexExtractionStrategy, CosineStrategy, JsonCssExtractionStrategy, and various chunking strategies with…", "heading": "Regex Extraction", "content": "Page: Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Regex Extraction\n\n```python\nimport json\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, RegexExtractionStrategy\n\n# Method 1: Use built-in patterns\nstrategy = RegexExtractionStrategy(\n    pattern =…\n```", "code_blocks": [{"language": "python", "code": "import json\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, RegexExtractionStrategy\n\n# Method 1: Use built-in patterns\nstrategy = RegexExtractionStrategy(\n    pattern =…", "filename": ""}], "chunk_position": 27, "heading_path": "Regex Extraction > Regex Extraction", "breadcrumbs": "Strategies - Crawl4AI Documentation (v0.9.x) > Regex Extraction > Regex Extraction"}, {"id": "28daa4ee4141c594", "url": "https://docs.crawl4ai.com/api/strategies/", "page_title": "Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "API reference for Crawl4AI extraction and chunking strategies, covering LLMExtractionStrategy, RegexExtractionStrategy, CosineStrategy, JsonCssExtractionStrategy, and various chunking strategies with…", "heading": "CSS Extraction", "content": "Page: Strategies - Crawl4AI Documentation (v0.9.x)\nSection: CSS Extraction\n\n```python\nfrom crawl4ai import JsonCssExtractionStrategy\n\n# Define schema\nschema = {\n    \"name\": \"Product List\",\n    \"baseSelector\": \".product-card\",\n    \"fields\": [\n        {\n            \"name\": \"title\",…\n```", "code_blocks": [{"language": "python", "code": "from crawl4ai import JsonCssExtractionStrategy\n\n# Define schema\nschema = {\n    \"name\": \"Product List\",\n    \"baseSelector\": \".product-card\",\n    \"fields\": [\n        {\n            \"name\": \"title\",…", "filename": ""}], "chunk_position": 27, "heading_path": "CSS Extraction > CSS Extraction", "breadcrumbs": "Strategies - Crawl4AI Documentation (v0.9.x) > CSS Extraction > CSS Extraction"}, {"id": "cb2aa2977fd2a065", "url": "https://docs.crawl4ai.com/api/strategies/", "page_title": "Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "API reference for Crawl4AI extraction and chunking strategies, covering LLMExtractionStrategy, RegexExtractionStrategy, CosineStrategy, JsonCssExtractionStrategy, and various chunking strategies with…", "heading": "Content Chunking", "content": "Page: Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Content Chunking\n\n```python\nfrom crawl4ai.chunking_strategy import OverlappingWindowChunking\nfrom crawl4ai import LLMConfig\n\n# Create chunking strategy\nchunker = OverlappingWindowChunking(\n    window_size=500,  # 500 words per…\n```", "code_blocks": [{"language": "python", "code": "from crawl4ai.chunking_strategy import OverlappingWindowChunking\nfrom crawl4ai import LLMConfig\n\n# Create chunking strategy\nchunker = OverlappingWindowChunking(\n    window_size=500,  # 500 words per…", "filename": ""}], "chunk_position": 27, "heading_path": "Content Chunking > Content Chunking", "breadcrumbs": "Strategies - Crawl4AI Documentation (v0.9.x) > Content Chunking > Content Chunking"}, {"id": "c3b93c359e7addb1", "url": "https://docs.crawl4ai.com/api/strategies/", "page_title": "Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "API reference for Crawl4AI extraction and chunking strategies, covering LLMExtractionStrategy, RegexExtractionStrategy, CosineStrategy, JsonCssExtractionStrategy, and various chunking strategies with…", "heading": "Choose the Right Strategy", "content": "Page: Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Choose the Right Strategy\n\n- Use `RegexExtractionStrategy` for common data types like emails, phones, URLs, dates\n- Use `JsonCssExtractionStrategy` for well-structured HTML with consistent patterns\n- Use…", "code_blocks": [], "chunk_position": 27, "heading_path": "Choose the Right Strategy > Choose the Right Strategy", "breadcrumbs": "Strategies - Crawl4AI Documentation (v0.9.x) > Choose the Right Strategy > Choose the Right Strategy"}, {"id": "5d2d62c9e2d33b68", "url": "https://docs.crawl4ai.com/api/strategies/", "page_title": "Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "API reference for Crawl4AI extraction and chunking strategies, covering LLMExtractionStrategy, RegexExtractionStrategy, CosineStrategy, JsonCssExtractionStrategy, and various chunking strategies with…", "heading": "Strategy Selection Guide", "content": "Page: Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Strategy Selection Guide\n\n```text\nIs the target data a common type (email/phone/date/URL)? \n→ RegexExtractionStrategy\n\nDoes the page have consistent HTML structure?\n→ JsonCssExtractionStrategy or JsonXPathExtractionStrategy\n\nIs the…\n```", "code_blocks": [{"language": "text", "code": "Is the target data a common type (email/phone/date/URL)? \n→ RegexExtractionStrategy\n\nDoes the page have consistent HTML structure?\n→ JsonCssExtractionStrategy or JsonXPathExtractionStrategy\n\nIs the…", "filename": ""}], "chunk_position": 27, "heading_path": "Strategy Selection Guide > Strategy Selection Guide", "breadcrumbs": "Strategies - Crawl4AI Documentation (v0.9.x) > Strategy Selection Guide > Strategy Selection Guide"}, {"id": "952037645e41bbc1", "url": "https://docs.crawl4ai.com/api/strategies/", "page_title": "Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "API reference for Crawl4AI extraction and chunking strategies, covering LLMExtractionStrategy, RegexExtractionStrategy, CosineStrategy, JsonCssExtractionStrategy, and various chunking strategies with…", "heading": "Optimize Chunking", "content": "Page: Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Optimize Chunking\n\n```python\n# For long documents\nstrategy = LLMExtractionStrategy(\n    chunk_token_threshold=2000,  # Smaller chunks\n    overlap_rate=0.1           # 10% overlap\n)\n```", "code_blocks": [{"language": "python", "code": "# For long documents\nstrategy = LLMExtractionStrategy(\n    chunk_token_threshold=2000,  # Smaller chunks\n    overlap_rate=0.1           # 10% overlap\n)", "filename": ""}], "chunk_position": 27, "heading_path": "Optimize Chunking > Optimize Chunking", "breadcrumbs": "Strategies - Crawl4AI Documentation (v0.9.x) > Optimize Chunking > Optimize Chunking"}, {"id": "4fec7adad4e30e0b", "url": "https://docs.crawl4ai.com/api/strategies/", "page_title": "Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "API reference for Crawl4AI extraction and chunking strategies, covering LLMExtractionStrategy, RegexExtractionStrategy, CosineStrategy, JsonCssExtractionStrategy, and various chunking strategies with…", "heading": "Combine Strategies for Best Performance", "content": "Page: Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Combine Strategies for Best Performance\n\n```python\n# First pass: Extract structure with CSS\ncss_strategy = JsonCssExtractionStrategy(product_schema)\ncss_result = await crawler.arun(url,…\n```", "code_blocks": [{"language": "python", "code": "# First pass: Extract structure with CSS\ncss_strategy = JsonCssExtractionStrategy(product_schema)\ncss_result = await crawler.arun(url,…", "filename": ""}], "chunk_position": 27, "heading_path": "Combine Strategies for Best Performance > Combine Strategies for Best Performance", "breadcrumbs": "Strategies - Crawl4AI Documentation (v0.9.x) > Combine Strategies for Best Performance > Combine Strategies for Best Performance"}, {"id": "f4db5fc09a06488d", "url": "https://docs.crawl4ai.com/api/strategies/", "page_title": "Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "API reference for Crawl4AI extraction and chunking strategies, covering LLMExtractionStrategy, RegexExtractionStrategy, CosineStrategy, JsonCssExtractionStrategy, and various chunking strategies with…", "heading": "Handle Errors", "content": "Page: Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Handle Errors\n\n```python\ntry:\n    result = await crawler.arun(\n        url=\"https://example.com\",\n        extraction_strategy=strategy\n    )\n    if result.success:\n        content =…\n```", "code_blocks": [{"language": "python", "code": "try:\n    result = await crawler.arun(\n        url=\"https://example.com\",\n        extraction_strategy=strategy\n    )\n    if result.success:\n        content =…", "filename": ""}], "chunk_position": 27, "heading_path": "Handle Errors > Handle Errors", "breadcrumbs": "Strategies - Crawl4AI Documentation (v0.9.x) > Handle Errors > Handle Errors"}, {"id": "b196aa42c07dc4f5", "url": "https://docs.crawl4ai.com/api/strategies/", "page_title": "Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "API reference for Crawl4AI extraction and chunking strategies, covering LLMExtractionStrategy, RegexExtractionStrategy, CosineStrategy, JsonCssExtractionStrategy, and various chunking strategies with…", "heading": "Monitor Performance", "content": "Page: Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Monitor Performance\n\n```python\nstrategy = CosineStrategy(\n    verbose=True,  # Enable logging\n    word_count_threshold=20,  # Filter short content\n    top_k=5  # Limit results\n)\n```", "code_blocks": [{"language": "python", "code": "strategy = CosineStrategy(\n    verbose=True,  # Enable logging\n    word_count_threshold=20,  # Filter short content\n    top_k=5  # Limit results\n)", "filename": ""}], "chunk_position": 27, "heading_path": "Monitor Performance > Monitor Performance", "breadcrumbs": "Strategies - Crawl4AI Documentation (v0.9.x) > Monitor Performance > Monitor Performance"}, {"id": "f05b65fba37968e2", "url": "https://docs.crawl4ai.com/api/strategies/", "page_title": "Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "API reference for Crawl4AI extraction and chunking strategies, covering LLMExtractionStrategy, RegexExtractionStrategy, CosineStrategy, JsonCssExtractionStrategy, and various chunking strategies with…", "heading": "Cache Generated Patterns", "content": "Page: Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Cache Generated Patterns\n\n```python\n# For RegexExtractionStrategy pattern generation\nimport json\nfrom pathlib import Path\n\ncache_dir = Path(\"./pattern_cache\")\ncache_dir.mkdir(exist_ok=True)\npattern_file = cache_dir /…\n```", "code_blocks": [{"language": "python", "code": "# For RegexExtractionStrategy pattern generation\nimport json\nfrom pathlib import Path\n\ncache_dir = Path(\"./pattern_cache\")\ncache_dir.mkdir(exist_ok=True)\npattern_file = cache_dir /…", "filename": ""}], "chunk_position": 27, "heading_path": "Cache Generated Patterns > Cache Generated Patterns", "breadcrumbs": "Strategies - Crawl4AI Documentation (v0.9.x) > Cache Generated Patterns > Cache Generated Patterns"}, {"id": "cfe3f43f08f72fb8", "url": "https://docs.crawl4ai.com/apps/", "page_title": "🚀 Crawl4AI Interactive Apps", "page_type": "overview", "page_summary": "Overview of Crawl4AI's interactive demo apps, including the C4A-Script editor, LLM context builder, Chrome extension assistant, and upcoming tools for scraping experiments, prompt design, and…", "heading": "🚀 Crawl4AI Interactive Apps", "content": "Page: 🚀 Crawl4AI Interactive Apps\nSection: 🚀 Crawl4AI Interactive Apps\n\nWelcome to the Crawl4AI Apps Hub - your gateway to interactive tools and demos that make web scraping more intuitive and powerful.", "code_blocks": [], "chunk_position": 28, "heading_path": "🚀 Crawl4AI Interactive Apps > 🚀 Crawl4AI Interactive Apps", "breadcrumbs": "🚀 Crawl4AI Interactive Apps > 🚀 Crawl4AI Interactive Apps > 🚀 Crawl4AI Interactive Apps"}, {"id": "09515c0d3972b460", "url": "https://docs.crawl4ai.com/apps/", "page_title": "🚀 Crawl4AI Interactive Apps", "page_type": "overview", "page_summary": "Overview of Crawl4AI's interactive demo apps, including the C4A-Script editor, LLM context builder, Chrome extension assistant, and upcoming tools for scraping experiments, prompt design, and…", "heading": "🛠️ Interactive Tools for Modern Web Scraping", "content": "Page: 🚀 Crawl4AI Interactive Apps\nSection: 🛠️ Interactive Tools for Modern Web Scraping\n\nOur apps are designed to make Crawl4AI more accessible and powerful. Whether you're learning browser automation, designing extraction strategies, or building complex scrapers, these tools provide…", "code_blocks": [], "chunk_position": 28, "heading_path": "🛠️ Interactive Tools for Modern Web Scraping > 🛠️ Interactive Tools for Modern Web Scraping", "breadcrumbs": "🚀 Crawl4AI Interactive Apps > 🛠️ Interactive Tools for Modern Web Scraping > 🛠️ Interactive Tools for Modern Web Scraping"}, {"id": "73547847e9fa773b", "url": "https://docs.crawl4ai.com/apps/", "page_title": "🚀 Crawl4AI Interactive Apps", "page_type": "overview", "page_summary": "Overview of Crawl4AI's interactive demo apps, including the C4A-Script editor, LLM context builder, Chrome extension assistant, and upcoming tools for scraping experiments, prompt design, and…", "heading": "🎨 C4A-Script Interactive Editor", "content": "Page: 🚀 Crawl4AI Interactive Apps\nSection: 🎨 C4A-Script Interactive Editor\n\nAvailable\n\nA visual, block-based programming environment for creating browser automation scripts. Perfect for beginners and experts alike!\n\n- Drag-and-drop visual programming\n- Real-time JavaScript…", "code_blocks": [], "chunk_position": 28, "heading_path": "🎨 C4A-Script Interactive Editor > 🎨 C4A-Script Interactive Editor", "breadcrumbs": "🚀 Crawl4AI Interactive Apps > 🎨 C4A-Script Interactive Editor > 🎨 C4A-Script Interactive Editor"}, {"id": "87ff1a14bb550578", "url": "https://docs.crawl4ai.com/apps/", "page_title": "🚀 Crawl4AI Interactive Apps", "page_type": "overview", "page_summary": "Overview of Crawl4AI's interactive demo apps, including the C4A-Script editor, LLM context builder, Chrome extension assistant, and upcoming tools for scraping experiments, prompt design, and…", "heading": "🧠 LLM Context Builder", "content": "Page: 🚀 Crawl4AI Interactive Apps\nSection: 🧠 LLM Context Builder\n\nAvailable\n\nGenerate optimized context files for your favorite LLM when working with Crawl4AI. Get focused, relevant documentation based on your needs.\n\n- Modular context generation\n- Memory,…", "code_blocks": [], "chunk_position": 28, "heading_path": "🧠 LLM Context Builder > 🧠 LLM Context Builder", "breadcrumbs": "🚀 Crawl4AI Interactive Apps > 🧠 LLM Context Builder > 🧠 LLM Context Builder"}, {"id": "c8c184fccff0f88c", "url": "https://docs.crawl4ai.com/apps/", "page_title": "🚀 Crawl4AI Interactive Apps", "page_type": "overview", "page_summary": "Overview of Crawl4AI's interactive demo apps, including the C4A-Script editor, LLM context builder, Chrome extension assistant, and upcoming tools for scraping experiments, prompt design, and…", "heading": "🕸️ Web Scraping Playground", "content": "Page: 🚀 Crawl4AI Interactive Apps\nSection: 🕸️ Web Scraping Playground\n\nComing Soon\n\nTest your scraping strategies on real websites with instant feedback. See how different configurations affect your results.\n\n- Live website testing\n- Side-by-side result comparison\n-…", "code_blocks": [], "chunk_position": 28, "heading_path": "🕸️ Web Scraping Playground > 🕸️ Web Scraping Playground", "breadcrumbs": "🚀 Crawl4AI Interactive Apps > 🕸️ Web Scraping Playground > 🕸️ Web Scraping Playground"}, {"id": "fbcf95bc811eb69d", "url": "https://docs.crawl4ai.com/apps/", "page_title": "🚀 Crawl4AI Interactive Apps", "page_type": "overview", "page_summary": "Overview of Crawl4AI's interactive demo apps, including the C4A-Script editor, LLM context builder, Chrome extension assistant, and upcoming tools for scraping experiments, prompt design, and…", "heading": "🔍 Crawl4AI Assistant (Chrome Extension)", "content": "Page: 🚀 Crawl4AI Interactive Apps\nSection: 🔍 Crawl4AI Assistant (Chrome Extension)\n\nAvailable\n\nVisual schema builder Chrome extension - click on webpage elements to generate extraction schemas and Python code!\n\n- Visual element selection\n- Container & field selection modes\n- Smart…", "code_blocks": [], "chunk_position": 28, "heading_path": "🔍 Crawl4AI Assistant (Chrome Extension) > 🔍 Crawl4AI Assistant (Chrome Extension)", "breadcrumbs": "🚀 Crawl4AI Interactive Apps > 🔍 Crawl4AI Assistant (Chrome Extension) > 🔍 Crawl4AI Assistant (Chrome Extension)"}, {"id": "c97bb09f06feb12d", "url": "https://docs.crawl4ai.com/apps/", "page_title": "🚀 Crawl4AI Interactive Apps", "page_type": "overview", "page_summary": "Overview of Crawl4AI's interactive demo apps, including the C4A-Script editor, LLM context builder, Chrome extension assistant, and upcoming tools for scraping experiments, prompt design, and…", "heading": "🧪 Extraction Lab", "content": "Page: 🚀 Crawl4AI Interactive Apps\nSection: 🧪 Extraction Lab\n\nComing Soon\n\nExperiment with different extraction strategies and see how they perform on your content. Compare LLM vs CSS vs XPath approaches.\n\n- Strategy comparison tools\n- Performance benchmarks\n-…", "code_blocks": [], "chunk_position": 28, "heading_path": "🧪 Extraction Lab > 🧪 Extraction Lab", "breadcrumbs": "🚀 Crawl4AI Interactive Apps > 🧪 Extraction Lab > 🧪 Extraction Lab"}, {"id": "d2b82e5eae399b18", "url": "https://docs.crawl4ai.com/apps/", "page_title": "🚀 Crawl4AI Interactive Apps", "page_type": "overview", "page_summary": "Overview of Crawl4AI's interactive demo apps, including the C4A-Script editor, LLM context builder, Chrome extension assistant, and upcoming tools for scraping experiments, prompt design, and…", "heading": "🤖 AI Prompt Designer", "content": "Page: 🚀 Crawl4AI Interactive Apps\nSection: 🤖 AI Prompt Designer\n\nComing Soon\n\nCraft and test prompts for LLM-based extraction. See how different prompts affect extraction quality and costs.\n\n- Prompt templates library\n- A/B testing interface\n- Token usage…", "code_blocks": [], "chunk_position": 28, "heading_path": "🤖 AI Prompt Designer > 🤖 AI Prompt Designer", "breadcrumbs": "🚀 Crawl4AI Interactive Apps > 🤖 AI Prompt Designer > 🤖 AI Prompt Designer"}, {"id": "50b2ffe3d89c4a8f", "url": "https://docs.crawl4ai.com/apps/", "page_title": "🚀 Crawl4AI Interactive Apps", "page_type": "overview", "page_summary": "Overview of Crawl4AI's interactive demo apps, including the C4A-Script editor, LLM context builder, Chrome extension assistant, and upcoming tools for scraping experiments, prompt design, and…", "heading": "📊 Crawl Monitor", "content": "Page: 🚀 Crawl4AI Interactive Apps\nSection: 📊 Crawl Monitor\n\nComing Soon\n\nReal-time monitoring dashboard for your crawling operations. Track performance, debug issues, and optimize your scrapers.\n\n- Real-time crawl statistics\n- Error tracking and debugging\n-…", "code_blocks": [], "chunk_position": 28, "heading_path": "📊 Crawl Monitor > 📊 Crawl Monitor", "breadcrumbs": "🚀 Crawl4AI Interactive Apps > 📊 Crawl Monitor > 📊 Crawl Monitor"}, {"id": "ce01b97f1658c132", "url": "https://docs.crawl4ai.com/apps/", "page_title": "🚀 Crawl4AI Interactive Apps", "page_type": "overview", "page_summary": "Overview of Crawl4AI's interactive demo apps, including the C4A-Script editor, LLM context builder, Chrome extension assistant, and upcoming tools for scraping experiments, prompt design, and…", "heading": "🎯 Accelerate Learning", "content": "Page: 🚀 Crawl4AI Interactive Apps\nSection: 🎯 Accelerate Learning\n\nVisual tools help you understand Crawl4AI's concepts faster than reading documentation alone.", "code_blocks": [], "chunk_position": 28, "heading_path": "🎯 Accelerate Learning > 🎯 Accelerate Learning", "breadcrumbs": "🚀 Crawl4AI Interactive Apps > 🎯 Accelerate Learning > 🎯 Accelerate Learning"}, {"id": "af16882600f08016", "url": "https://docs.crawl4ai.com/apps/", "page_title": "🚀 Crawl4AI Interactive Apps", "page_type": "overview", "page_summary": "Overview of Crawl4AI's interactive demo apps, including the C4A-Script editor, LLM context builder, Chrome extension assistant, and upcoming tools for scraping experiments, prompt design, and…", "heading": "💡 Reduce Development Time", "content": "Page: 🚀 Crawl4AI Interactive Apps\nSection: 💡 Reduce Development Time\n\nGenerate working code instantly instead of writing everything from scratch.", "code_blocks": [], "chunk_position": 28, "heading_path": "💡 Reduce Development Time > 💡 Reduce Development Time", "breadcrumbs": "🚀 Crawl4AI Interactive Apps > 💡 Reduce Development Time > 💡 Reduce Development Time"}, {"id": "f4ea03ee4a99a574", "url": "https://docs.crawl4ai.com/apps/", "page_title": "🚀 Crawl4AI Interactive Apps", "page_type": "overview", "page_summary": "Overview of Crawl4AI's interactive demo apps, including the C4A-Script editor, LLM context builder, Chrome extension assistant, and upcoming tools for scraping experiments, prompt design, and…", "heading": "🔍 Improve Quality", "content": "Page: 🚀 Crawl4AI Interactive Apps\nSection: 🔍 Improve Quality\n\nTest and refine your approach before deploying to production.", "code_blocks": [], "chunk_position": 28, "heading_path": "🔍 Improve Quality > 🔍 Improve Quality", "breadcrumbs": "🚀 Crawl4AI Interactive Apps > 🔍 Improve Quality > 🔍 Improve Quality"}, {"id": "a12ac427ff00a99c", "url": "https://docs.crawl4ai.com/apps/", "page_title": "🚀 Crawl4AI Interactive Apps", "page_type": "overview", "page_summary": "Overview of Crawl4AI's interactive demo apps, including the C4A-Script editor, LLM context builder, Chrome extension assistant, and upcoming tools for scraping experiments, prompt design, and…", "heading": "🤝 Community Driven", "content": "Page: 🚀 Crawl4AI Interactive Apps\nSection: 🤝 Community Driven\n\nThese tools are built based on user feedback. Have an idea? [Let us know](https://github.com/unclecode/crawl4ai/issues)!", "code_blocks": [], "chunk_position": 28, "heading_path": "🤝 Community Driven > 🤝 Community Driven", "breadcrumbs": "🚀 Crawl4AI Interactive Apps > 🤝 Community Driven > 🤝 Community Driven"}, {"id": "c3039f0b9c04eccf", "url": "https://docs.crawl4ai.com/apps/", "page_title": "🚀 Crawl4AI Interactive Apps", "page_type": "overview", "page_summary": "Overview of Crawl4AI's interactive demo apps, including the C4A-Script editor, LLM context builder, Chrome extension assistant, and upcoming tools for scraping experiments, prompt design, and…", "heading": "📢 Stay Updated", "content": "Page: 🚀 Crawl4AI Interactive Apps\nSection: 📢 Stay Updated\n\nWant to know when new apps are released?\n\n- ⭐ [Star us on GitHub](https://github.com/unclecode/crawl4ai) to get notifications\n- 🐦 Follow [@unclecode](https://twitter.com/unclecode) for…", "code_blocks": [], "chunk_position": 28, "heading_path": "📢 Stay Updated > 📢 Stay Updated", "breadcrumbs": "🚀 Crawl4AI Interactive Apps > 📢 Stay Updated > 📢 Stay Updated"}, {"id": "e83142377bc12995", "url": "https://docs.crawl4ai.com/apps/", "page_title": "🚀 Crawl4AI Interactive Apps", "page_type": "overview", "page_summary": "Overview of Crawl4AI's interactive demo apps, including the C4A-Script editor, LLM context builder, Chrome extension assistant, and upcoming tools for scraping experiments, prompt design, and…", "heading": "Developer Resources", "content": "Page: 🚀 Crawl4AI Interactive Apps\nSection: Developer Resources\n\nBuilding your own tools with Crawl4AI? Check out our [API Reference](../api/async-webcrawler/) and [Integration Guide](../advanced/advanced-features/) for comprehensive documentation.", "code_blocks": [], "chunk_position": 28, "heading_path": "Developer Resources > Developer Resources", "breadcrumbs": "🚀 Crawl4AI Interactive Apps > Developer Resources > Developer Resources"}, {"id": "1d693f9481a547f2", "url": "https://docs.crawl4ai.com/apps/llmtxt/build/", "page_title": "Build - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page contains a detailed prompt for an AI coding assistant to build an interactive HTML/JavaScript page that lets users select and combine crawl4ai LLM context Markdown files into a single…", "heading": "Objective", "content": "Page: Build - Crawl4AI Documentation (v0.9.x)\nSection: Objective\n\nYour task is to create an interactive HTML webpage with JavaScript functionality that allows users to select and combine different `crawl4ai` LLM context files into a single downloadable Markdown…", "code_blocks": [], "chunk_position": 29, "heading_path": "Objective > Objective", "breadcrumbs": "Build - Crawl4AI Documentation (v0.9.x) > Objective > Objective"}, {"id": "a3a5f5e4d61bc359", "url": "https://docs.crawl4ai.com/apps/llmtxt/build/", "page_title": "Build - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page contains a detailed prompt for an AI coding assistant to build an interactive HTML/JavaScript page that lets users select and combine crawl4ai LLM context Markdown files into a single…", "heading": "Core Functionality", "content": "Page: Build - Crawl4AI Documentation (v0.9.x)\nSection: Core Functionality\n\n- **Display `crawl4ai` Components:** The page will list all available `crawl4ai` documentation components.\n- **Select Context Types:** For each component, users can select which types of context they…", "code_blocks": [], "chunk_position": 29, "heading_path": "Core Functionality > Core Functionality", "breadcrumbs": "Build - Crawl4AI Documentation (v0.9.x) > Core Functionality > Core Functionality"}, {"id": "33b69ca5d64fe94f", "url": "https://docs.crawl4ai.com/apps/llmtxt/build/", "page_title": "Build - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page contains a detailed prompt for an AI coding assistant to build an interactive HTML/JavaScript page that lets users select and combine crawl4ai LLM context Markdown files into a single…", "heading": "Input/Assumptions", "content": "Page: Build - Crawl4AI Documentation (v0.9.x)\nSection: Input/Assumptions\n\n- **Context Files Location:** All individual context Markdown files are located on the server in a publicly accessible folder named `llmtxt/`.\n- **File Naming Convention:** Files follow the pattern:…", "code_blocks": [], "chunk_position": 29, "heading_path": "Input/Assumptions > Input/Assumptions", "breadcrumbs": "Build - Crawl4AI Documentation (v0.9.x) > Input/Assumptions > Input/Assumptions"}, {"id": "ba65be03b691705e", "url": "https://docs.crawl4ai.com/apps/llmtxt/build/", "page_title": "Build - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page contains a detailed prompt for an AI coding assistant to build an interactive HTML/JavaScript page that lets users select and combine crawl4ai LLM context Markdown files into a single…", "heading": "Detailed UI/UX Requirements", "content": "Page: Build - Crawl4AI Documentation (v0.9.x)\nSection: Detailed UI/UX Requirements\n\n- **Main Page Structure:**\n  - **Header:** \"Crawl4AI Interactive LLM Context Builder\"\n  - **Introduction:** Briefly explain the purpose of the tool (from the `USING_LLM_CONTEXTS.md` content you…", "code_blocks": [], "chunk_position": 29, "heading_path": "Detailed UI/UX Requirements > Detailed UI/UX Requirements", "breadcrumbs": "Build - Crawl4AI Documentation (v0.9.x) > Detailed UI/UX Requirements > Detailed UI/UX Requirements"}, {"id": "047a3e4dd775194c", "url": "https://docs.crawl4ai.com/apps/llmtxt/build/", "page_title": "Build - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page contains a detailed prompt for an AI coding assistant to build an interactive HTML/JavaScript page that lets users select and combine crawl4ai LLM context Markdown files into a single…", "heading": "Final Output", "content": "Page: Build - Crawl4AI Documentation (v0.9.x)\nSection: Final Output\n\n- A single HTML file (e.g., `interactive_context_builder.html`).\n- Associated JavaScript code (can be inline within `<script>` tags or in a separate `.js` file).\n- Associated CSS code (can be inline…", "code_blocks": [], "chunk_position": 29, "heading_path": "Final Output > Final Output", "breadcrumbs": "Build - Crawl4AI Documentation (v0.9.x) > Final Output > Final Output"}, {"id": "44c38dcaaf42b844", "url": "https://docs.crawl4ai.com/apps/llmtxt/why/", "page_title": "Supercharging Your AI Assistant: My Journey to Better LLM Contexts for `crawl4ai`", "page_type": "guide", "page_summary": "Explains the limitations of standard llm.txt files for providing AI coding assistants with context for crawl4ai, and introduces a multi-dimensional, modular context system using memory, reasoning,…", "heading": "Introduction", "content": "Page: Supercharging Your AI Assistant: My Journey to Better LLM Contexts for `crawl4ai`\nSection: Introduction\n\nWhen I started diving deep into using AI coding assistants with my own libraries, particularly `crawl4ai`, I quickly realized that the common approach to providing context via a simple `llm.txt` or…", "code_blocks": [], "chunk_position": 30, "heading_path": "Introduction > Introduction", "breadcrumbs": "Supercharging Your AI Assistant: My Journey to Better LLM Contexts for `crawl4ai` > Introduction > Introduction"}, {"id": "90c5eaf36deb5205", "url": "https://docs.crawl4ai.com/apps/llmtxt/why/", "page_title": "Supercharging Your AI Assistant: My Journey to Better LLM Contexts for `crawl4ai`", "page_type": "guide", "page_summary": "Explains the limitations of standard llm.txt files for providing AI coding assistants with context for crawl4ai, and introduces a multi-dimensional, modular context system using memory, reasoning,…", "heading": "My Frustration with Standard `llm.txt` Files", "content": "Page: Supercharging Your AI Assistant: My Journey to Better LLM Contexts for `crawl4ai`\nSection: My Frustration with Standard `llm.txt` Files\n\nMy experience with generic `llm.txt` files for complex libraries like `crawl4ai` revealed several pain points:\n\n- **Information Overload & Lost Focus:** I found that when I threw a massive,…", "code_blocks": [], "chunk_position": 30, "heading_path": "My Frustration with Standard `llm.txt` Files > My Frustration with Standard `llm.txt` Files", "breadcrumbs": "Supercharging Your AI Assistant: My Journey to Better LLM Contexts for `crawl4ai` > My Frustration with Standard `llm.txt` Files > My Frustration with Standard `llm.txt` Files"}, {"id": "8b5b079a4b392d34", "url": "https://docs.crawl4ai.com/apps/llmtxt/why/", "page_title": "Supercharging Your AI Assistant: My Journey to Better LLM Contexts for `crawl4ai`", "page_type": "guide", "page_summary": "Explains the limitations of standard llm.txt files for providing AI coding assistants with context for crawl4ai, and introduces a multi-dimensional, modular context system using memory, reasoning,…", "heading": "Inspiration: Selective Inclusion & Multi-Dimensional Understanding", "content": "Page: Supercharging Your AI Assistant: My Journey to Better LLM Contexts for `crawl4ai`\nSection: Inspiration: Selective Inclusion & Multi-Dimensional Understanding\n\nI've always admired how libraries like Lodash or jQuery (in its modular days) allowed developers to pick and choose only the parts they needed, resulting in smaller, more focused bundles. This idea…", "code_blocks": [], "chunk_position": 30, "heading_path": "Inspiration: Selective Inclusion & Multi-Dimensional Understanding > Inspiration: Selective Inclusion & Multi-Dimensional Understanding", "breadcrumbs": "Supercharging Your AI Assistant: My Journey to Better LLM Contexts for `crawl4ai` > Inspiration: Selective Inclusion & Multi-Dimensional Understanding > Inspiration: Selective Inclusion & Multi-Dimensional Understanding"}, {"id": "ab62f37e08c317bd", "url": "https://docs.crawl4ai.com/basic/installation/", "page_title": "Installation 💻 - Crawl4AI Documentation (v0.9.x)", "page_type": "reference", "page_summary": "Extraction fallback content.", "heading": "Installation 💻 - Crawl4AI Documentation (v0.9.x)", "content": "Page: Installation 💻 - Crawl4AI Documentation (v0.9.x)\nSection: Installation 💻 - Crawl4AI Documentation (v0.9.x)\n\n\n\n\n##### Installation 💻\n\n\n\n\nCrawl4AI offers flexible installation options to suit various use cases. You can install it as a Python package, use it with Docker, or run it as a local server.\n\n\n\n\n##### Option 1: Python Package Installation (Recommended)\n\n\n\n\nCrawl4AI is now available on PyPI, making installation easier than ever. Choose the option that best fits your needs:\n\n\n\n\n###### Basic Installation\n\n\n\n\nFor basic web crawling and scraping tasks:\n\n\n\n\n\n###### Installation with PyTorch\n\n\n\n\nFor advanced text clustering (includes CosineSimilarity cluster strategy):\n\n\n\n\n\n###### Installation with Transformers\n\n\n\n\nFor text summarization and Hugging Face models:\n\n\n\n\n\n###### Full Installation\n\n\n\n\nFor all features:\n\n\n\n\n\n###### Development Installation\n\n\n\n\nFor contributors who plan to modify the source code:\n\n\n\n\n\n💡 After installation with \"torch\", \"transformer\", or \"all\" options, it's recommended to run the following CLI command to load the required models:\n\n\n\n\n\nThis is optional but will boost the performance and speed of the crawler. You only need to do this once after installation.\n\n\n\n\n##### Playwright Installation Note for Ubuntu\n\n\n\n\nIf you encounter issues with Playwright installation on Ubuntu, you may need to install additional dependencies:\n\n\n\n\n\n##### Option 2: Using Docker (Coming Soon)\n\n\n\n\nDocker support for Crawl4AI is currently in progress and will be available soon. This will allow you to run Crawl4AI in a containerized environment, ensuring consistency across different systems.\n\n\n\n\n##### Option 3: Local Server Installation\n\n\n\n\nFor those who prefer to run Crawl4AI as a local server, instructions will be provided once the Docker implementation is complete.\n\n\n\n\n##### Verifying Your Installation\n\n\n\n\nAfter installation, you can verify that Crawl4AI is working correctly by running a simple Python script:\n\n\n\n\n\nThis script should successfully crawl the example website and print the first 500 characters of the extracted content.\n\n\n\n\n##### Getting Help\n\n\n\n\nIf you encounter any issues during installation or usage, please check the [documentation](https://docs.crawl4ai.com/) or raise an issue on the [GitHub repository](https://github.com/unclecode/crawl4ai/issues).\n\n\n\n\nHappy crawling! 🕷️🤖\n\n\n\nPage Copy\nPage Copy\n\n\n\n\n- [Copy as Markdown\nCopy page for LLMs](#)\n\n- [View as Markdown\nOpen raw source](#)\n\n\n- [Open in ChatGPT\nAsk questions about this page](#)\n\n\n\nESC to close\n", "code_blocks": [{"language": "bash", "code": "pip install crawl4ai\nplaywright install # Install Playwright dependencies", "filename": ""}, {"language": "css", "code": "pip install crawl4ai[torch]", "filename": ""}, {"language": "css", "code": "pip install crawl4ai[transformer]", "filename": ""}, {"language": "css", "code": "pip install crawl4ai[all]", "filename": ""}, {"language": "bash", "code": "git clone https://github.com/unclecode/crawl4ai.git\ncd crawl4ai\npip install -e \".[all]\"\nplaywright install # Install Playwright dependencies", "filename": ""}, {"language": "undefined", "code": "crawl4ai-download-models", "filename": ""}, {"language": "csharp", "code": "sudo apt-get install -y \\\n    libwoff1 \\\n    libopus0 \\\n    libwebp7 \\\n    libwebpdemux2 \\\n    libenchant-2-2 \\\n    libgudev-1.0-0 \\\n    libsecret-1-0 \\\n    libhyphen0 \\\n    libgdk-pixbuf2.0-0 \\\n    libegl1 \\\n    libnotify4 \\\n    libxslt1.1 \\\n    libevent-2.1-7 \\\n    libgles2 \\\n    libxcomposite1 \\\n    libatk1.0-0 \\\n    libatk-bridge2.0-0 \\\n    libepoxy0 \\\n    libgtk-3-0 \\\n    libharfbuzz-icu0 \\\n    libgstreamer-gl1.0-0 \\\n    libgstreamer-plugins-bad1.0-0 \\\n    gstreamer1.0-plugins-good \\\n    gstreamer1.0-plugins-bad \\\n    libxt6 \\\n    libxaw7 \\\n    xvfb \\\n    fonts-noto-color-emoji \\\n    libfontconfig \\\n    libfreetype6 \\\n    xfonts-cyrillic \\\n    xfonts-scalable \\\n    fonts-liberation \\\n    fonts-ipafont-gothic \\\n    fonts-wqy-zenhei \\\n    fonts-tlwg-loma-otf \\\n    fonts-freefont-ttf", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler\n\nasync def main():\n    async with AsyncWebCrawler(verbose=True) as crawler:\n        result = await crawler.arun(url=\"https://www.example.com\")\n        print(result.markdown[:500])  # Print first 500 characters\n\nif __name__ == \"__main__\":\n    asyncio.run(main())", "filename": ""}], "chunk_position": 31, "heading_path": "Installation 💻 - Crawl4AI Documentation (v0.9.x) > Installation 💻 - Crawl4AI Documentation (v0.9.x)", "breadcrumbs": "Installation 💻 - Crawl4AI Documentation (v0.9.x) > Installation 💻 - Crawl4AI Documentation (v0.9.x) > Installation 💻 - Crawl4AI Documentation (v0.9.x)"}, {"id": "d8c68029c5352ad9", "url": "https://docs.crawl4ai.com/complete-sdk-reference/", "page_title": "Crawl4AI Complete SDK Documentation", "page_type": "reference", "page_summary": "Comprehensive SDK reference for Crawl4AI, covering installation, quick start, core API (AsyncWebCrawler, arun, arun_many, CrawlResult), configuration, and extraction strategies.", "heading": "Installation & Setup", "content": "Page: Crawl4AI Complete SDK Documentation\nSection: Installation & Setup\n\n##### Installation & Setup (2023 Edition)\n\n##### 1. Basic Installation\n\n##### 2. Initial Setup & Diagnostics\n\n###### 2.1 Run the Setup Command\n\n- Performs OS-level checks (e.g., missing libs on Linux)\n- Confirms…", "code_blocks": [{"language": "bash", "code": "pip install crawl4ai", "filename": ""}, {"language": "bash", "code": "crawl4ai-setup", "filename": ""}, {"language": "bash", "code": "crawl4ai-doctor", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig\n\nasync def main():\n    async with AsyncWebCrawler() as crawler:\n        result = await crawler.arun(…", "filename": ""}, {"language": "bash", "code": "pip install crawl4ai[torch]\ncrawl4ai-setup", "filename": ""}, {"language": "bash", "code": "pip install crawl4ai[transformer]\ncrawl4ai-setup", "filename": ""}, {"language": "bash", "code": "pip install crawl4ai[all]\ncrawl4ai-setup", "filename": ""}, {"language": "bash", "code": "crawl4ai-download-models", "filename": ""}, {"language": "bash", "code": "docker pull unclecode/crawl4ai:basic\ndocker run -p 11235:11235 unclecode/crawl4ai:basic", "filename": ""}], "chunk_position": 32, "heading_path": "Installation & Setup > Installation & Setup", "breadcrumbs": "Crawl4AI Complete SDK Documentation > Installation & Setup > Installation & Setup"}, {"id": "4fdc5d1f9fd6b8ad", "url": "https://docs.crawl4ai.com/complete-sdk-reference/", "page_title": "Crawl4AI Complete SDK Documentation", "page_type": "reference", "page_summary": "Comprehensive SDK reference for Crawl4AI, covering installation, quick start, core API (AsyncWebCrawler, arun, arun_many, CrawlResult), configuration, and extraction strategies.", "heading": "Quick Start", "content": "Page: Crawl4AI Complete SDK Documentation\nSection: Quick Start\n\n##### Getting Started with Crawl4AI\n\n- Run your **first crawl** using minimal configuration.\n- Experiment with a simple **CSS-based extraction** strategy.\n- Crawl a **dynamic** page that loads content…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler\n\nasync def main():\n    async with AsyncWebCrawler() as crawler:\n        result = await crawler.arun(\"https://example.com\")…", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode\n\nasync def main():\n    browser_conf = BrowserConfig(headless=True)  # or False to see the browser…", "filename": ""}, {"language": "python", "code": "from crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai.content_filter_strategy import PruningContentFilter\nfrom crawl4ai.markdown_generation_strategy import…", "filename": ""}, {"language": "python", "code": "from crawl4ai import JsonCssExtractionStrategy\nfrom crawl4ai import LLMConfig\n\n# Generate a schema (one-time cost)\nhtml = \"<div class='product'><h2>Gaming Laptop</h2><span…", "filename": ""}, {"language": "python", "code": "import asyncio\nimport json\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\nfrom crawl4ai import JsonCssExtractionStrategy\n\nasync def main():\n    schema = {\n        \"name\": \"Example…", "filename": ""}, {"language": "python", "code": "import os\nimport json\nimport asyncio\nfrom pydantic import BaseModel, Field\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, LLMConfig\nfrom crawl4ai import LLMExtractionStrategy\n\nclass…", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, AdaptiveCrawler\n\nasync def adaptive_example():\n    async with AsyncWebCrawler() as crawler:\n        adaptive = AdaptiveCrawler(crawler)\n\n        #…", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\n\nasync def quick_parallel_example():\n    urls = [\n        \"https://example.com/page1\",…", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode\nfrom crawl4ai import JsonCssExtractionStrategy\n\nasync def…", "filename": ""}], "chunk_position": 32, "heading_path": "Quick Start > Quick Start", "breadcrumbs": "Crawl4AI Complete SDK Documentation > Quick Start > Quick Start"}, {"id": "0d5be54a5cac5440", "url": "https://docs.crawl4ai.com/complete-sdk-reference/", "page_title": "Crawl4AI Complete SDK Documentation", "page_type": "reference", "page_summary": "Comprehensive SDK reference for Crawl4AI, covering installation, quick start, core API (AsyncWebCrawler, arun, arun_many, CrawlResult), configuration, and extraction strategies.", "heading": "Core API", "content": "Page: Crawl4AI Complete SDK Documentation\nSection: Core API\n\n##### AsyncWebCrawler\n\nThe **`AsyncWebCrawler`** is the core class for asynchronous web crawling in Crawl4AI. You typically create it **once**, optionally customize it with a **`BrowserConfig`** (e.g.,…", "code_blocks": [{"language": "python", "code": "class AsyncWebCrawler:\n    def __init__(\n        self,\n        crawler_strategy: Optional[AsyncCrawlerStrategy] = None,\n        config: Optional[BrowserConfig] = None,\n        always_bypass_cache:…", "filename": ""}, {"language": "python", "code": "from crawl4ai import AsyncWebCrawler, BrowserConfig\nbrowser_cfg = BrowserConfig(\n    browser_type=\"chromium\",\n    headless=True,\n    verbose=True\n)\ncrawler = AsyncWebCrawler(config=browser_cfg)", "filename": ""}, {"language": "python", "code": "async with AsyncWebCrawler(config=browser_cfg) as crawler:\n    result = await crawler.arun(\"https://example.com\")\n    # The crawler automatically starts/closes resources", "filename": ""}, {"language": "python", "code": "crawler = AsyncWebCrawler(config=browser_cfg)\nawait crawler.start()\nresult1 = await crawler.arun(\"https://example.com\")\nresult2 = await crawler.arun(\"https://another.com\")\nawait crawler.close()", "filename": ""}, {"language": "python", "code": "async def arun(\n    url: str,\n    config: Optional[CrawlerRunConfig] = None,\n    # Legacy parameters for backward compatibility...", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import CrawlerRunConfig, CacheMode\nrun_cfg = CrawlerRunConfig(\n    cache_mode=CacheMode.BYPASS,\n    css_selector=\"main.article\",\n    word_count_threshold=10,…", "filename": ""}, {"language": "python", "code": "async def arun_many(\n    urls: List[str],\n    config: Optional[CrawlerRunConfig] = None,\n    # Legacy parameters maintained for backwards compatibility...", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode\nfrom crawl4ai import JsonCssExtractionStrategy\nimport json\n\nasync def main():\n    # 1. Browser config…", "filename": ""}, {"language": "python", "code": "run_cfg = CrawlerRunConfig(css_selector=\".main-content\", word_count_threshold=20)\nresult = await crawler.arun(url=\"...\", config=run_cfg)", "filename": ""}], "chunk_position": 32, "heading_path": "Core API > Core API", "breadcrumbs": "Crawl4AI Complete SDK Documentation > Core API > Core API"}, {"id": "1034db494ab79733", "url": "https://docs.crawl4ai.com/complete-sdk-reference/", "page_title": "Crawl4AI Complete SDK Documentation", "page_type": "reference", "page_summary": "Comprehensive SDK reference for Crawl4AI, covering installation, quick start, core API (AsyncWebCrawler, arun, arun_many, CrawlResult), configuration, and extraction strategies.", "heading": "`arun()` Parameter Guide (New Approach)", "content": "Page: Crawl4AI Complete SDK Documentation\nSection: `arun()` Parameter Guide (New Approach)\n\nIn Crawl4AI’s **latest** configuration model, nearly all parameters that once went directly to `arun()` are now part of **`CrawlerRunConfig`**. When calling `arun()`, you provide:\n\nBelow is an…", "code_blocks": [{"language": "python", "code": "await crawler.arun(\n    url=\"https://example.com\",  \n    config=my_run_config\n)", "filename": ""}, {"language": "python", "code": "from crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\n\nasync def main():\n    run_config = CrawlerRunConfig(\n        verbose=True,            # Detailed logging…", "filename": ""}, {"language": "python", "code": "run_config = CrawlerRunConfig(\n    cache_mode=CacheMode.BYPASS\n)", "filename": ""}, {"language": "python", "code": "run_config = CrawlerRunConfig(\n    word_count_threshold=10,   # Ignore text blocks <10 words\n    only_text=False,           # If True, tries to remove non-text elements\n    keep_data_attributes=False…", "filename": ""}, {"language": "python", "code": "run_config = CrawlerRunConfig(\n    css_selector=\".main-content\",  # Focus on .main-content region only\n    excluded_tags=[\"form\", \"nav\"], # Remove entire tag blocks\n    remove_forms=True,…", "filename": ""}, {"language": "python", "code": "run_config = CrawlerRunConfig(\n    exclude_external_links=True,         # Remove external links from final content\n    exclude_social_media_links=True,     # Remove links to known social sites…", "filename": ""}, {"language": "python", "code": "run_config = CrawlerRunConfig(\n    exclude_external_images=True  # Strip images from other domains\n)", "filename": ""}, {"language": "python", "code": "run_config = CrawlerRunConfig(\n    wait_for=\"css:.dynamic-content\", # Wait for .dynamic-content\n    delay_before_return_html=2.0,    # Wait 2s before capturing final HTML\n    page_timeout=60000,…", "filename": ""}, {"language": "python", "code": "run_config = CrawlerRunConfig(\n    js_code=[\n        \"window.scrollTo(0, document.body.scrollHeight);\",\n        \"document.querySelector('.load-more')?.click();\"\n    ],\n    js_only=False\n)", "filename": ""}, {"language": "python", "code": "run_config = CrawlerRunConfig(\n    magic=True,\n    simulate_user=True,\n    override_navigator=True\n)", "filename": ""}, {"language": "python", "code": "run_config = CrawlerRunConfig(\n    session_id=\"my_session123\"\n)", "filename": ""}, {"language": "python", "code": "run_config = CrawlerRunConfig(\n    screenshot=True,             # Grab a screenshot as base64\n    screenshot_wait_for=1.0,     # Wait 1s before capturing\n    pdf=True,                    # Also…", "filename": ""}, {"language": "python", "code": "run_config = CrawlerRunConfig(\n    extraction_strategy=my_css_or_llm_strategy\n)", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\nfrom crawl4ai import JsonCssExtractionStrategy\n\nasync def main():\n    # Example schema\n    schema = {\n        \"name\":…", "filename": ""}], "chunk_position": 32, "heading_path": "`arun()` Parameter Guide (New Approach) > `arun()` Parameter Guide (New Approach)", "breadcrumbs": "Crawl4AI Complete SDK Documentation > `arun()` Parameter Guide (New Approach) > `arun()` Parameter Guide (New Approach)"}, {"id": "63b7d54bfa30cf70", "url": "https://docs.crawl4ai.com/complete-sdk-reference/", "page_title": "Crawl4AI Complete SDK Documentation", "page_type": "reference", "page_summary": "Comprehensive SDK reference for Crawl4AI, covering installation, quick start, core API (AsyncWebCrawler, arun, arun_many, CrawlResult), configuration, and extraction strategies.", "heading": "`arun_many(...)` Reference", "content": "Page: Crawl4AI Complete SDK Documentation\nSection: `arun_many(...)` Reference\n\n> **Note**: This function is very similar to [`arun()`](./arun.md) but focused on **concurrent** or **batch** crawling. If you’re unfamiliar with `arun()` usage, please read that doc first, then…", "code_blocks": [{"language": "python", "code": "async def arun_many(\n    urls: Union[List[str], List[Any]],\n    config: Optional[Union[CrawlerRunConfig, List[CrawlerRunConfig]]] = None,\n    dispatcher: Optional[BaseDispatcher] = None,\n    ...\n) ->…", "filename": ""}, {"language": "python", "code": "# Minimal usage: The default dispatcher will be used\nresults = await crawler.arun_many(\n    urls=[\"https://site1.com\", \"https://site2.com\"],\n    config=CrawlerRunConfig(stream=False)  # Default…", "filename": ""}, {"language": "python", "code": "config = CrawlerRunConfig(\n    stream=True,  # Enable streaming mode\n    cache_mode=CacheMode.BYPASS\n)\n\n# Process results as they complete\nasync for result in await crawler.arun_many(…", "filename": ""}, {"language": "python", "code": "dispatcher = MemoryAdaptiveDispatcher(\n    memory_threshold_percent=70.0,\n    max_session_permit=10\n)\nresults = await crawler.arun_many(\n    urls=[\"https://site1.com\", \"https://site2.com\",…", "filename": ""}, {"language": "python", "code": "from crawl4ai import CrawlerRunConfig, MatchMode\nfrom crawl4ai.processors.pdf import PDFContentScrapingStrategy\nfrom crawl4ai.extraction_strategy import JsonCssExtractionStrategy\nfrom…", "filename": ""}], "chunk_position": 32, "heading_path": "`arun_many(...)` Reference > `arun_many(...)` Reference", "breadcrumbs": "Crawl4AI Complete SDK Documentation > `arun_many(...)` Reference > `arun_many(...)` Reference"}, {"id": "cb68fce33a995203", "url": "https://docs.crawl4ai.com/complete-sdk-reference/", "page_title": "Crawl4AI Complete SDK Documentation", "page_type": "reference", "page_summary": "Comprehensive SDK reference for Crawl4AI, covering installation, quick start, core API (AsyncWebCrawler, arun, arun_many, CrawlResult), configuration, and extraction strategies.", "heading": "`CrawlResult` Reference", "content": "Page: Crawl4AI Complete SDK Documentation\nSection: `CrawlResult` Reference\n\nThe **`CrawlResult`** class encapsulates everything returned after a single crawl operation. It provides the **raw or processed content**, details on links and media, plus optional metadata (like…", "code_blocks": [{"language": "python", "code": "class CrawlResult(BaseModel):\n    url: str\n    html: str\n    success: bool\n    cleaned_html: Optional[str] = None\n    fit_html: Optional[str] = None  # Preprocessed HTML optimized for extraction…", "filename": ""}, {"language": "python", "code": "print(result.url)  # e.g., \"https://example.com/\"", "filename": ""}, {"language": "python", "code": "if not result.success:\n    print(f\"Crawl failed: {result.error_message}\")", "filename": ""}, {"language": "python", "code": "if result.status_code == 404:\n    print(\"Page not found!\")", "filename": ""}, {"language": "python", "code": "if not result.success:\n    print(\"Error:\", result.error_message)", "filename": ""}, {"language": "python", "code": "# If you used session_id=\"login_session\" in CrawlerRunConfig, see it here:\nprint(\"Session:\", result.session_id)", "filename": ""}, {"language": "python", "code": "if result.response_headers:\n    print(\"Server:\", result.response_headers.get(\"Server\", \"Unknown\"))", "filename": ""}, {"language": "python", "code": "if result.ssl_certificate:\n    print(\"Issuer:\", result.ssl_certificate.issuer)", "filename": ""}, {"language": "python", "code": "# Possibly large\nprint(len(result.html))", "filename": ""}, {"language": "python", "code": "print(result.cleaned_html[:500])  # Show a snippet", "filename": ""}, {"language": "python", "code": "if result.markdown:\n    md_res = result.markdown\n    print(\"Raw MD:\", md_res.raw_markdown[:300])\n    print(\"Citations MD:\", md_res.markdown_with_citations[:300])\n    print(\"References:\",…", "filename": ""}, {"language": "python", "code": "print(result.markdown.raw_markdown[:200])\nprint(result.markdown.fit_markdown)\nprint(result.markdown.fit_html)", "filename": ""}, {"language": "python", "code": "images = result.media.get(\"images\", [])\nfor img in images:\n    if img.get(\"score\", 0) > 5:\n        print(\"High-value image:\", img[\"src\"])", "filename": ""}, {"language": "python", "code": "for link in result.links[\"internal\"]:\n    print(f\"Internal link to {link['href']} with text {link['text']}\")", "filename": ""}, {"language": "python", "code": "if result.extracted_content:\n    data = json.loads(result.extracted_content)\n    print(data)", "filename": ""}, {"language": "python", "code": "if result.downloaded_files:\n    for file_path in result.downloaded_files:\n        print(\"Downloaded:\", file_path)", "filename": ""}, {"language": "python", "code": "import base64\nif result.screenshot:\n    with open(\"page.png\", \"wb\") as f:\n        f.write(base64.b64decode(result.screenshot))", "filename": ""}, {"language": "python", "code": "if result.pdf:\n    with open(\"page.pdf\", \"wb\") as f:\n        f.write(result.pdf)", "filename": ""}, {"language": "python", "code": "if result.mhtml:\n    with open(\"page.mhtml\", \"w\", encoding=\"utf-8\") as f:\n        f.write(result.mhtml)", "filename": ""}, {"language": "python", "code": "if result.metadata:\n    print(\"Title:\", result.metadata.get(\"title\"))\n    print(\"Author:\", result.metadata.get(\"author\"))", "filename": ""}, {"language": "python", "code": "# Example usage:\nfor result in results:\n    if result.success and result.dispatch_result:\n        dr = result.dispatch_result\n        print(f\"URL: {result.url}, Task ID: {dr.task_id}\")…", "filename": ""}], "chunk_position": 32, "heading_path": "`CrawlResult` Reference > `CrawlResult` Reference", "breadcrumbs": "Crawl4AI Complete SDK Documentation > `CrawlResult` Reference > `CrawlResult` Reference"}, {"id": "658e00756db92b7e", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Introduction", "content": "Page: Adaptive Web Crawling\nSection: Introduction\n\nTraditional web crawlers follow predetermined patterns, crawling pages blindly without knowing when they've gathered enough information. **Adaptive Crawling** changes this paradigm by introducing…", "code_blocks": [], "chunk_position": 33, "heading_path": "Introduction > Introduction", "breadcrumbs": "Adaptive Web Crawling > Introduction > Introduction"}, {"id": "eb3bb1223ac0f604", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "The Problem It Solves", "content": "Page: Adaptive Web Crawling\nSection: The Problem It Solves\n\nWhen crawling websites for specific information, you face two challenges:\n1. **Under-crawling**: Stopping too early and missing crucial information\n2. **Over-crawling**: Wasting resources by crawling…", "code_blocks": [], "chunk_position": 33, "heading_path": "The Problem It Solves > The Problem It Solves", "breadcrumbs": "Adaptive Web Crawling > The Problem It Solves > The Problem It Solves"}, {"id": "79a0ad6787ebeacb", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "How It Works", "content": "Page: Adaptive Web Crawling\nSection: How It Works\n\nThe AdaptiveCrawler uses three metrics to measure information sufficiency:\n\n- **Coverage**: How well your collected pages cover the query terms\n- **Consistency**: Whether the information is coherent…", "code_blocks": [], "chunk_position": 33, "heading_path": "How It Works > How It Works", "breadcrumbs": "Adaptive Web Crawling > How It Works > How It Works"}, {"id": "6673389dd424d588", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Basic Usage", "content": "Page: Adaptive Web Crawling\nSection: Basic Usage\n\n```\nfrom crawl4ai import AsyncWebCrawler, AdaptiveCrawler\n\nasync def main():\n    async with AsyncWebCrawler() as crawler:\n        # Create an adaptive crawler (config is optional)\n        adaptive =…\n```", "code_blocks": [{"language": "", "code": "from crawl4ai import AsyncWebCrawler, AdaptiveCrawler\n\nasync def main():\n    async with AsyncWebCrawler() as crawler:\n        # Create an adaptive crawler (config is optional)\n        adaptive =…", "filename": ""}], "chunk_position": 33, "heading_path": "Basic Usage > Basic Usage", "breadcrumbs": "Adaptive Web Crawling > Basic Usage > Basic Usage"}, {"id": "bdc496f8f73f006e", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Configuration Options", "content": "Page: Adaptive Web Crawling\nSection: Configuration Options\n\n```\nfrom crawl4ai import AdaptiveConfig\n\nconfig = AdaptiveConfig(\n    confidence_threshold=0.8,    # Stop when 80% confident (default: 0.7)\n    max_pages=30,               # Maximum pages to crawl…\n```", "code_blocks": [{"language": "", "code": "from crawl4ai import AdaptiveConfig\n\nconfig = AdaptiveConfig(\n    confidence_threshold=0.8,    # Stop when 80% confident (default: 0.7)\n    max_pages=30,               # Maximum pages to crawl…", "filename": ""}], "chunk_position": 33, "heading_path": "Configuration Options > Configuration Options", "breadcrumbs": "Adaptive Web Crawling > Configuration Options > Configuration Options"}, {"id": "f6c1205c52de2407", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Crawling Strategies", "content": "Page: Adaptive Web Crawling\nSection: Crawling Strategies\n\nAdaptive Crawling supports two distinct strategies for determining information sufficiency:", "code_blocks": [], "chunk_position": 33, "heading_path": "Crawling Strategies > Crawling Strategies", "breadcrumbs": "Adaptive Web Crawling > Crawling Strategies > Crawling Strategies"}, {"id": "21f7b989ed272a26", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Statistical Strategy (Default)", "content": "Page: Adaptive Web Crawling\nSection: Statistical Strategy (Default)\n\nThe statistical strategy uses pure information theory and term-based analysis:\n\n- **Fast and efficient** - No API calls or model loading\n- **Term-based coverage** - Analyzes query term presence and…", "code_blocks": [{"language": "", "code": "# Default configuration uses statistical strategy\nconfig = AdaptiveConfig(\n    strategy=\"statistical\",  # This is the default\n    confidence_threshold=0.8\n)", "filename": ""}], "chunk_position": 33, "heading_path": "Statistical Strategy (Default) > Statistical Strategy (Default)", "breadcrumbs": "Adaptive Web Crawling > Statistical Strategy (Default) > Statistical Strategy (Default)"}, {"id": "27aabc0cf4dc0106", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Embedding Strategy", "content": "Page: Adaptive Web Crawling\nSection: Embedding Strategy\n\nThe embedding strategy uses semantic embeddings for deeper understanding:\n\n- **Semantic understanding** - Captures meaning beyond exact term matches\n- **Query expansion** - Automatically generates…", "code_blocks": [{"language": "", "code": "# Configure embedding strategy with local embeddings\nconfig = AdaptiveConfig(\n    strategy=\"embedding\",\n    embedding_model=\"sentence-transformers/all-MiniLM-L6-v2\",  # Default…", "filename": ""}], "chunk_position": 33, "heading_path": "Embedding Strategy > Embedding Strategy", "breadcrumbs": "Adaptive Web Crawling > Embedding Strategy > Embedding Strategy"}, {"id": "af40075eb38c1ae2", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Strategy Comparison", "content": "Page: Adaptive Web Crawling\nSection: Strategy Comparison\n\n| Feature | Statistical | Embedding |\n| --- | --- | --- |\n| **Speed** | Very fast | Moderate (API calls) |\n| **Cost** | Free | Depends on provider |\n| **Accuracy** | Good for exact terms | Excellent…", "code_blocks": [], "chunk_position": 33, "heading_path": "Strategy Comparison > Strategy Comparison", "breadcrumbs": "Adaptive Web Crawling > Strategy Comparison > Strategy Comparison"}, {"id": "e67509f75f311522", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Embedding Strategy Configuration", "content": "Page: Adaptive Web Crawling\nSection: Embedding Strategy Configuration\n\n```\nconfig = AdaptiveConfig(\n    strategy=\"embedding\",\n\n    # Model configuration\n    embedding_model=\"sentence-transformers/all-MiniLM-L6-v2\",\n    embedding_llm_config=None,  # Use for API-based…\n```", "code_blocks": [{"language": "", "code": "config = AdaptiveConfig(\n    strategy=\"embedding\",\n\n    # Model configuration\n    embedding_model=\"sentence-transformers/all-MiniLM-L6-v2\",\n    embedding_llm_config=None,  # Use for API-based…", "filename": ""}], "chunk_position": 33, "heading_path": "Embedding Strategy Configuration > Embedding Strategy Configuration", "breadcrumbs": "Adaptive Web Crawling > Embedding Strategy Configuration > Embedding Strategy Configuration"}, {"id": "b95aac511d9d792f", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Handling Irrelevant Queries", "content": "Page: Adaptive Web Crawling\nSection: Handling Irrelevant Queries\n\nThe embedding strategy can detect when a query is completely unrelated to the content:", "code_blocks": [{"language": "", "code": "# This will stop quickly with low confidence\nresult = await adaptive.digest(\n    start_url=\"https://docs.python.org/3/\",\n    query=\"how to cook pasta\"  # Irrelevant to Python docs\n)\n\n# Check if query…", "filename": ""}], "chunk_position": 33, "heading_path": "Handling Irrelevant Queries > Handling Irrelevant Queries", "breadcrumbs": "Adaptive Web Crawling > Handling Irrelevant Queries > Handling Irrelevant Queries"}, {"id": "ca5bffaec23f2a95", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Perfect For:", "content": "Page: Adaptive Web Crawling\nSection: Perfect For:\n\n- **Research Tasks**: Finding comprehensive information about a topic\n- **Question Answering**: Gathering sufficient context to answer specific queries\n- **Knowledge Base Building**: Creating focused…", "code_blocks": [], "chunk_position": 33, "heading_path": "Perfect For: > Perfect For:", "breadcrumbs": "Adaptive Web Crawling > Perfect For: > Perfect For:"}, {"id": "4d5ac547e44ecfb6", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Not Recommended For:", "content": "Page: Adaptive Web Crawling\nSection: Not Recommended For:\n\n- **Full Site Archiving**: When you need every page regardless of content\n- **Structured Data Extraction**: When targeting specific, known page patterns\n- **Real-time Monitoring**: When you need…", "code_blocks": [], "chunk_position": 33, "heading_path": "Not Recommended For: > Not Recommended For:", "breadcrumbs": "Adaptive Web Crawling > Not Recommended For: > Not Recommended For:"}, {"id": "4dad773f9b23b115", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Confidence Score", "content": "Page: Adaptive Web Crawling\nSection: Confidence Score\n\nThe confidence score (0-1) indicates how sufficient the gathered information is:\n- **0.0-0.3**: Insufficient information, needs more crawling\n- **0.3-0.6**: Partial information, may answer basic…", "code_blocks": [], "chunk_position": 33, "heading_path": "Confidence Score > Confidence Score", "breadcrumbs": "Adaptive Web Crawling > Confidence Score > Confidence Score"}, {"id": "0824c62c6e788a84", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Statistics Display", "content": "Page: Adaptive Web Crawling\nSection: Statistics Display\n\nThe summary shows:\n- Pages crawled vs. confidence achieved\n- Coverage, consistency, and saturation scores\n- Crawling efficiency metrics", "code_blocks": [{"language": "", "code": "adaptive.print_stats(detailed=False)  # Summary table\nadaptive.print_stats(detailed=True)   # Detailed metrics", "filename": ""}], "chunk_position": 33, "heading_path": "Statistics Display > Statistics Display", "breadcrumbs": "Adaptive Web Crawling > Statistics Display > Statistics Display"}, {"id": "bc26cf9daf40eea9", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Saving Progress", "content": "Page: Adaptive Web Crawling\nSection: Saving Progress\n\n```\nconfig = AdaptiveConfig(\n    save_state=True,\n    state_path=\"my_crawl_state.json\"\n)\n\n# Crawl will auto-save progress\nresult = await adaptive.digest(start_url, query)\n```", "code_blocks": [{"language": "", "code": "config = AdaptiveConfig(\n    save_state=True,\n    state_path=\"my_crawl_state.json\"\n)\n\n# Crawl will auto-save progress\nresult = await adaptive.digest(start_url, query)", "filename": ""}], "chunk_position": 33, "heading_path": "Saving Progress > Saving Progress", "breadcrumbs": "Adaptive Web Crawling > Saving Progress > Saving Progress"}, {"id": "bc6b06c4243cd0f6", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Resuming a Crawl", "content": "Page: Adaptive Web Crawling\nSection: Resuming a Crawl\n\n```\n# Resume from saved state\nresult = await adaptive.digest(\n    start_url,\n    query,\n    resume_from=\"my_crawl_state.json\"\n)\n```", "code_blocks": [{"language": "", "code": "# Resume from saved state\nresult = await adaptive.digest(\n    start_url,\n    query,\n    resume_from=\"my_crawl_state.json\"\n)", "filename": ""}], "chunk_position": 33, "heading_path": "Resuming a Crawl > Resuming a Crawl", "breadcrumbs": "Adaptive Web Crawling > Resuming a Crawl > Resuming a Crawl"}, {"id": "5fa0ec8411124c04", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Exporting Knowledge Base", "content": "Page: Adaptive Web Crawling\nSection: Exporting Knowledge Base\n\n```\n# Export collected pages to JSONL\nadaptive.export_knowledge_base(\"knowledge_base.jsonl\")\n\n# Import into another session\nnew_adaptive = AdaptiveCrawler(crawler)\nawait…\n```", "code_blocks": [{"language": "", "code": "# Export collected pages to JSONL\nadaptive.export_knowledge_base(\"knowledge_base.jsonl\")\n\n# Import into another session\nnew_adaptive = AdaptiveCrawler(crawler)\nawait…", "filename": ""}], "chunk_position": 33, "heading_path": "Exporting Knowledge Base > Exporting Knowledge Base", "breadcrumbs": "Adaptive Web Crawling > Exporting Knowledge Base > Exporting Knowledge Base"}, {"id": "81dc247d33758d82", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "1. Query Formulation", "content": "Page: Adaptive Web Crawling\nSection: 1. Query Formulation\n\n- Use specific, descriptive queries\n- Include key terms you expect to find\n- Avoid overly broad queries", "code_blocks": [], "chunk_position": 33, "heading_path": "1. Query Formulation > 1. Query Formulation", "breadcrumbs": "Adaptive Web Crawling > 1. Query Formulation > 1. Query Formulation"}, {"id": "3ae941ed47cbb688", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "2. Threshold Tuning", "content": "Page: Adaptive Web Crawling\nSection: 2. Threshold Tuning\n\n- Start with default (0.7) for general use\n- Lower to 0.5-0.6 for exploratory crawling\n- Raise to 0.8+ for exhaustive coverage", "code_blocks": [], "chunk_position": 33, "heading_path": "2. Threshold Tuning > 2. Threshold Tuning", "breadcrumbs": "Adaptive Web Crawling > 2. Threshold Tuning > 2. Threshold Tuning"}, {"id": "554dad7849f1327f", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "3. Performance Optimization", "content": "Page: Adaptive Web Crawling\nSection: 3. Performance Optimization\n\n- Use appropriate `max_pages` limits\n- Adjust `top_k_links` based on site structure\n- Enable caching for repeat crawls", "code_blocks": [], "chunk_position": 33, "heading_path": "3. Performance Optimization > 3. Performance Optimization", "breadcrumbs": "Adaptive Web Crawling > 3. Performance Optimization > 3. Performance Optimization"}, {"id": "e62c6c850cb71cb4", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "4. Link Selection", "content": "Page: Adaptive Web Crawling\nSection: 4. Link Selection\n\n- The crawler prioritizes links based on:\n- Relevance to query\n- Expected information gain\n- URL structure and depth", "code_blocks": [], "chunk_position": 33, "heading_path": "4. Link Selection > 4. Link Selection", "breadcrumbs": "Adaptive Web Crawling > 4. Link Selection > 4. Link Selection"}, {"id": "3b6f84ac1463ee7c", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Research Assistant", "content": "Page: Adaptive Web Crawling\nSection: Research Assistant\n\n```\n# Gather information about a programming concept\nresult = await adaptive.digest(\n    start_url=\"https://realpython.com\",\n    query=\"python decorators implementation patterns\"\n)\n\n# Get the most…\n```", "code_blocks": [{"language": "", "code": "# Gather information about a programming concept\nresult = await adaptive.digest(\n    start_url=\"https://realpython.com\",\n    query=\"python decorators implementation patterns\"\n)\n\n# Get the most…", "filename": ""}], "chunk_position": 33, "heading_path": "Research Assistant > Research Assistant", "breadcrumbs": "Adaptive Web Crawling > Research Assistant > Research Assistant"}, {"id": "fcc093886d50ab32", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Knowledge Base Builder", "content": "Page: Adaptive Web Crawling\nSection: Knowledge Base Builder\n\n```\n# Build a focused knowledge base about machine learning\nqueries = [\n    \"supervised learning algorithms\",\n    \"neural network architectures\",\n    \"model evaluation metrics\"\n]\n\nfor query in queries:…\n```", "code_blocks": [{"language": "", "code": "# Build a focused knowledge base about machine learning\nqueries = [\n    \"supervised learning algorithms\",\n    \"neural network architectures\",\n    \"model evaluation metrics\"\n]\n\nfor query in queries:…", "filename": ""}], "chunk_position": 33, "heading_path": "Knowledge Base Builder > Knowledge Base Builder", "breadcrumbs": "Adaptive Web Crawling > Knowledge Base Builder > Knowledge Base Builder"}, {"id": "804ffe3dfef44296", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "API Documentation Crawler", "content": "Page: Adaptive Web Crawling\nSection: API Documentation Crawler\n\n```\n# Intelligently crawl API documentation\nconfig = AdaptiveConfig(\n    confidence_threshold=0.85,  # Higher threshold for completeness\n    max_pages=30\n)\n\nadaptive = AdaptiveCrawler(crawler,…\n```", "code_blocks": [{"language": "", "code": "# Intelligently crawl API documentation\nconfig = AdaptiveConfig(\n    confidence_threshold=0.85,  # Higher threshold for completeness\n    max_pages=30\n)\n\nadaptive = AdaptiveCrawler(crawler,…", "filename": ""}], "chunk_position": 33, "heading_path": "API Documentation Crawler > API Documentation Crawler", "breadcrumbs": "Adaptive Web Crawling > API Documentation Crawler > API Documentation Crawler"}, {"id": "76a8907b7bbe0254", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "Next Steps", "content": "Page: Adaptive Web Crawling\nSection: Next Steps\n\n- Learn about [Advanced Adaptive Strategies](../../advanced/adaptive-strategies/)\n- Explore the [AdaptiveCrawler API Reference](../../api/adaptive-crawler/)\n- See more…", "code_blocks": [], "chunk_position": 33, "heading_path": "Next Steps > Next Steps", "breadcrumbs": "Adaptive Web Crawling > Next Steps > Next Steps"}, {"id": "885ae697a31fadb3", "url": "https://docs.crawl4ai.com/core/adaptive-crawling/", "page_title": "Adaptive Web Crawling", "page_type": "guide", "page_summary": "Adaptive Web Crawling is a Crawl4AI guide to intelligent crawling, using coverage, consistency, and saturation metrics to decide when enough information has been collected. It covers configuration,…", "heading": "FAQ", "content": "Page: Adaptive Web Crawling\nSection: FAQ\n\n**Q: How is this different from traditional crawling?**\nA: Traditional crawling follows fixed patterns (BFS/DFS). Adaptive crawling makes intelligent decisions about which links to follow and when to…", "code_blocks": [], "chunk_position": 33, "heading_path": "FAQ > FAQ", "breadcrumbs": "Adaptive Web Crawling > FAQ > FAQ"}, {"id": "279f571a919950a1", "url": "https://docs.crawl4ai.com/core/ask-ai/", "page_title": "Ask AI - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the Ask AI feature in Crawl4AI, which enables users to ask questions about crawled web content using LLM providers. It covers the ask_ai method, LLM configuration, and usage…", "heading": "Overview", "content": "Page: Ask AI - Crawl4AI Documentation (v0.9.x)\nSection: Overview\n\nThe Ask AI feature in Crawl4AI allows you to ask questions about the content you have crawled. Instead of manually parsing and analyzing the extracted content, you can leverage Large Language Models…", "code_blocks": [], "chunk_position": 34, "heading_path": "Overview > Overview", "breadcrumbs": "Ask AI - Crawl4AI Documentation (v0.9.x) > Overview > Overview"}, {"id": "2d1283ba56a4494e", "url": "https://docs.crawl4ai.com/core/ask-ai/", "page_title": "Ask AI - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the Ask AI feature in Crawl4AI, which enables users to ask questions about crawled web content using LLM providers. It covers the ask_ai method, LLM configuration, and usage…", "heading": "Basic Usage", "content": "Page: Ask AI - Crawl4AI Documentation (v0.9.x)\nSection: Basic Usage\n\nTo use Ask AI, you first need to crawl a page and then pass the extracted content along with your question to the ask_ai method. The method requires an LLMConfig object that specifies which LLM…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, LLMConfig\n\nasync def main():\n    llm_config = LLMConfig(provider=\"openai/gpt-4o\", api_token=\"your-api-token\")\n    async with AsyncWebCrawler() as…", "filename": ""}], "chunk_position": 34, "heading_path": "Basic Usage > Basic Usage", "breadcrumbs": "Ask AI - Crawl4AI Documentation (v0.9.x) > Basic Usage > Basic Usage"}, {"id": "0fb11f5ccfc6d2dc", "url": "https://docs.crawl4ai.com/core/ask-ai/", "page_title": "Ask AI - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the Ask AI feature in Crawl4AI, which enables users to ask questions about crawled web content using LLM providers. It covers the ask_ai method, LLM configuration, and usage…", "heading": "LLM Configuration", "content": "Page: Ask AI - Crawl4AI Documentation (v0.9.x)\nSection: LLM Configuration\n\nThe LLMConfig class is used to configure the LLM provider for Ask AI. It supports a wide range of providers through the litellm library, including OpenAI, Anthropic, Google Gemini, Azure OpenAI, and…", "code_blocks": [{"language": "python", "code": "from crawl4ai import LLMConfig\n\n# OpenAI\nllm_config = LLMConfig(\n    provider=\"openai/gpt-4o\",\n    api_token=\"your-openai-api-token\"\n)\n\n# Anthropic\nllm_config = LLMConfig(…", "filename": ""}], "chunk_position": 34, "heading_path": "LLM Configuration > LLM Configuration", "breadcrumbs": "Ask AI - Crawl4AI Documentation (v0.9.x) > LLM Configuration > LLM Configuration"}, {"id": "1e0c9fa83f8d1853", "url": "https://docs.crawl4ai.com/core/ask-ai/", "page_title": "Ask AI - Crawl4AI Documentation (v0.9.x)", "page_type": "api", "page_summary": "This page documents the Ask AI feature in Crawl4AI, which enables users to ask questions about crawled web content using LLM providers. It covers the ask_ai method, LLM configuration, and usage…", "heading": "Advanced Usage", "content": "Page: Ask AI - Crawl4AI Documentation (v0.9.x)\nSection: Advanced Usage\n\nYou can customize the behavior of the LLM by providing a system prompt, adjusting the temperature, and setting the maximum number of tokens in the response. This allows you to tailor the AI's…", "code_blocks": [{"language": "python", "code": "answer = await crawler.ask_ai(\n    question=\"Extract all product names and prices from this page.\",\n    context=result.markdown,\n    llm_config=llm_config,\n    system_prompt=\"You are a helpful…", "filename": ""}], "chunk_position": 34, "heading_path": "Advanced Usage > Advanced Usage", "breadcrumbs": "Ask AI - Crawl4AI Documentation (v0.9.x) > Advanced Usage > Advanced Usage"}, {"id": "4e4235f98f28dd41", "url": "https://docs.crawl4ai.com/core/browser-crawler-config/", "page_title": "Browser, Crawler & LLM Configuration (Quick Overview)", "page_type": "guide", "page_summary": "Overview of the three core configuration classes in Crawl4AI — BrowserConfig, CrawlerRunConfig, and LLMConfig — explaining their most commonly used parameters, helper methods, and how to combine them…", "heading": "Overview", "content": "Page: Browser, Crawler & LLM Configuration (Quick Overview)\nSection: Overview\n\nCrawl4AI's flexibility stems from two key classes:\n\n- **`BrowserConfig`**  – Dictates  **how**  the browser is launched and behaves (e.g., headless or visible, proxy, user agent).\n\n-…", "code_blocks": [], "chunk_position": 35, "heading_path": "Overview > Overview", "breadcrumbs": "Browser, Crawler & LLM Configuration (Quick Overview) > Overview > Overview"}, {"id": "d5d9138a12814e26", "url": "https://docs.crawl4ai.com/core/browser-crawler-config/", "page_title": "Browser, Crawler & LLM Configuration (Quick Overview)", "page_type": "guide", "page_summary": "Overview of the three core configuration classes in Crawl4AI — BrowserConfig, CrawlerRunConfig, and LLMConfig — explaining their most commonly used parameters, helper methods, and how to combine them…", "heading": "1. BrowserConfig Essentials", "content": "Page: Browser, Crawler & LLM Configuration (Quick Overview)\nSection: 1. BrowserConfig Essentials\n\n###### Key Fields to Note\n\n1.⠀ **`browser_type`** \n\n   - Options: `\"chromium\"`, `\"firefox\"`, or `\"webkit\"`.\n\n   - Defaults to `\"chromium\"`.\n\n   - If you need a different engine, specify it here.\n\n2.⠀…", "code_blocks": [{"language": "python", "code": "class BrowserConfig:\n    def __init__(\n        browser_type=\"chromium\",\n        headless=True,\n        browser_mode=\"dedicated\",\n        use_managed_browser=False,\n        cdp_url=None,…", "filename": ""}, {"language": "json", "code": "{\n    \"server\": \"http://proxy.example.com:8080\", \n    \"username\": \"...\", \n    \"password\": \"...\"\n}", "filename": ""}, {"language": "python", "code": "# Create a base browser config\nbase_browser = BrowserConfig(\n    browser_type=\"chromium\",\n    headless=True,\n    text_mode=True\n)\n\n# Create a visible browser config for debugging\ndebug_browser =…", "filename": ""}, {"language": "python", "code": "from crawl4ai import BrowserConfig, CrawlerRunConfig\n\n# At application startup — one time\nBrowserConfig.set_defaults(\n    cache_cdp_connection=True,\n    cdp_close_delay=0,…", "filename": ""}, {"language": "python", "code": "from crawl4ai import AsyncWebCrawler, BrowserConfig\n\nbrowser_conf = BrowserConfig(\n    browser_type=\"firefox\",\n    headless=False,\n    text_mode=True\n)\n\nasync with…", "filename": ""}], "chunk_position": 35, "heading_path": "1. BrowserConfig Essentials > 1. BrowserConfig Essentials", "breadcrumbs": "Browser, Crawler & LLM Configuration (Quick Overview) > 1. BrowserConfig Essentials > 1. BrowserConfig Essentials"}, {"id": "47967657498f5d4e", "url": "https://docs.crawl4ai.com/core/browser-crawler-config/", "page_title": "Browser, Crawler & LLM Configuration (Quick Overview)", "page_type": "guide", "page_summary": "Overview of the three core configuration classes in Crawl4AI — BrowserConfig, CrawlerRunConfig, and LLMConfig — explaining their most commonly used parameters, helper methods, and how to combine them…", "heading": "2. CrawlerRunConfig Essentials", "content": "Page: Browser, Crawler & LLM Configuration (Quick Overview)\nSection: 2. CrawlerRunConfig Essentials\n\n###### Key Fields to Note\n\n1.⠀ **`word_count_threshold`** :\n\n   - The minimum word count before a block is considered.\n\n   - If your site has lots of short paragraphs or items, you can lower it.\n\n2.⠀…", "code_blocks": [{"language": "python", "code": "class CrawlerRunConfig:\n    def __init__(\n        word_count_threshold=200,\n        extraction_strategy=None,\n        chunking_strategy=RegexChunking(),\n        markdown_generator=None,…", "filename": ""}, {"language": "python", "code": "# Create a base configuration\nbase_config = CrawlerRunConfig(\n    cache_mode=CacheMode.ENABLED,\n    word_count_threshold=200,\n    wait_until=\"networkidle\"\n)\n\n# Create variations for different use…", "filename": ""}], "chunk_position": 35, "heading_path": "2. CrawlerRunConfig Essentials > 2. CrawlerRunConfig Essentials", "breadcrumbs": "Browser, Crawler & LLM Configuration (Quick Overview) > 2. CrawlerRunConfig Essentials > 2. CrawlerRunConfig Essentials"}, {"id": "b06e6747e4a999df", "url": "https://docs.crawl4ai.com/core/browser-crawler-config/", "page_title": "Browser, Crawler & LLM Configuration (Quick Overview)", "page_type": "guide", "page_summary": "Overview of the three core configuration classes in Crawl4AI — BrowserConfig, CrawlerRunConfig, and LLMConfig — explaining their most commonly used parameters, helper methods, and how to combine them…", "heading": "3. LLMConfig Essentials", "content": "Page: Browser, Crawler & LLM Configuration (Quick Overview)\nSection: 3. LLMConfig Essentials\n\n###### Key fields to note\n\n1.⠀ **`provider`** :\n\n- Which LLM provider to use. \n- Possible values are `\"ollama/llama3\",\"groq/llama3-70b-8192\",\"groq/llama3-8b-8192\", \"openai/gpt-4o-mini\"…", "code_blocks": [{"language": "python", "code": "llm_config = LLMConfig(\n    provider=\"openai/gpt-4o-mini\",\n    api_token=os.getenv(\"OPENAI_API_KEY\"),\n    backoff_base_delay=1, # optional\n    backoff_max_attempts=5, # optional…", "filename": ""}], "chunk_position": 35, "heading_path": "3. LLMConfig Essentials > 3. LLMConfig Essentials", "breadcrumbs": "Browser, Crawler & LLM Configuration (Quick Overview) > 3. LLMConfig Essentials > 3. LLMConfig Essentials"}, {"id": "63f0caa112c33ec2", "url": "https://docs.crawl4ai.com/core/browser-crawler-config/", "page_title": "Browser, Crawler & LLM Configuration (Quick Overview)", "page_type": "guide", "page_summary": "Overview of the three core configuration classes in Crawl4AI — BrowserConfig, CrawlerRunConfig, and LLMConfig — explaining their most commonly used parameters, helper methods, and how to combine them…", "heading": "4. Putting It All Together", "content": "Page: Browser, Crawler & LLM Configuration (Quick Overview)\nSection: 4. Putting It All Together\n\nIn a typical scenario, you define  **one**  `BrowserConfig` for your crawler session, then create  **one or more**  `CrawlerRunConfig` & `LLMConfig` depending on each call's needs:", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode, LLMConfig, LLMContentFilter, DefaultMarkdownGenerator\nfrom crawl4ai import…", "filename": ""}], "chunk_position": 35, "heading_path": "4. Putting It All Together > 4. Putting It All Together", "breadcrumbs": "Browser, Crawler & LLM Configuration (Quick Overview) > 4. Putting It All Together > 4. Putting It All Together"}, {"id": "f22c949f60eec254", "url": "https://docs.crawl4ai.com/core/browser-crawler-config/", "page_title": "Browser, Crawler & LLM Configuration (Quick Overview)", "page_type": "guide", "page_summary": "Overview of the three core configuration classes in Crawl4AI — BrowserConfig, CrawlerRunConfig, and LLMConfig — explaining their most commonly used parameters, helper methods, and how to combine them…", "heading": "5. Next Steps", "content": "Page: Browser, Crawler & LLM Configuration (Quick Overview)\nSection: 5. Next Steps\n\nFor a  **detailed list**  of available parameters (including advanced ones), see:\n\n- [BrowserConfig, CrawlerRunConfig & LLMConfig Reference](../../api/parameters/)\n\nYou can explore topics like:\n\n-…", "code_blocks": [], "chunk_position": 35, "heading_path": "5. Next Steps > 5. Next Steps", "breadcrumbs": "Browser, Crawler & LLM Configuration (Quick Overview) > 5. Next Steps > 5. Next Steps"}, {"id": "5dac75feb81c87f6", "url": "https://docs.crawl4ai.com/core/browser-crawler-config/", "page_title": "Browser, Crawler & LLM Configuration (Quick Overview)", "page_type": "guide", "page_summary": "Overview of the three core configuration classes in Crawl4AI — BrowserConfig, CrawlerRunConfig, and LLMConfig — explaining their most commonly used parameters, helper methods, and how to combine them…", "heading": "6. Conclusion", "content": "Page: Browser, Crawler & LLM Configuration (Quick Overview)\nSection: 6. Conclusion\n\n**BrowserConfig** ,  **CrawlerRunConfig**  and  **LLMConfig**  give you straightforward ways to define:\n\n- **Which**  browser to launch, how it should run, and any proxy or user agent needs.\n-…", "code_blocks": [], "chunk_position": 35, "heading_path": "6. Conclusion > 6. Conclusion", "breadcrumbs": "Browser, Crawler & LLM Configuration (Quick Overview) > 6. Conclusion > 6. Conclusion"}, {"id": "e579ce174b8ac63e", "url": "https://docs.crawl4ai.com/core/c4a-script/", "page_title": "C4A-Script - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "A comprehensive guide to C4A-Script, a human-readable DSL for web automation, covering syntax, commands, examples, and advanced features.", "heading": "What is C4A-Script?", "content": "Page: C4A-Script - Crawl4AI Documentation (v0.9.x)\nSection: What is C4A-Script?\n\nC4A-Script is a powerful, human-readable domain-specific language (DSL) designed for web automation and interaction. Think of it as a simplified programming language that anyone can read and write,…", "code_blocks": [{"language": "text", "code": "# Navigate and interact in plain English\nGO https://example.com\nWAIT `#search-box` 5\nTYPE \"Hello World\"\nCLICK `button[type=\"submit\"]`\nCopy", "filename": ""}], "chunk_position": 36, "heading_path": "What is C4A-Script? > What is C4A-Script?", "breadcrumbs": "C4A-Script - Crawl4AI Documentation (v0.9.x) > What is C4A-Script? > What is C4A-Script?"}, {"id": "5cc4f32f927475ae", "url": "https://docs.crawl4ai.com/core/c4a-script/", "page_title": "C4A-Script - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "A comprehensive guide to C4A-Script, a human-readable DSL for web automation, covering syntax, commands, examples, and advanced features.", "heading": "Getting Started: Your First Script", "content": "Page: C4A-Script - Crawl4AI Documentation (v0.9.x)\nSection: Getting Started: Your First Script\n\nLet's create a simple script that searches for something on a website:\n\nThat's it! In just a few lines, you've automated a complete search workflow.", "code_blocks": [{"language": "text", "code": "# My first C4A-Script\nGO https://duckduckgo.com\n\n# Wait for the search box to appear\nWAIT `input[name=\"q\"]` 10\n\n# Type our search query\nTYPE \"Crawl4AI\"\n\n# Press Enter to search\nPRESS Enter\n\n# Wait…", "filename": ""}], "chunk_position": 36, "heading_path": "Getting Started: Your First Script > Getting Started: Your First Script", "breadcrumbs": "C4A-Script - Crawl4AI Documentation (v0.9.x) > Getting Started: Your First Script > Getting Started: Your First Script"}, {"id": "1879cc86acfebdb7", "url": "https://docs.crawl4ai.com/core/c4a-script/", "page_title": "C4A-Script - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "A comprehensive guide to C4A-Script, a human-readable DSL for web automation, covering syntax, commands, examples, and advanced features.", "heading": "Interactive Tutorial & Live Demo", "content": "Page: C4A-Script - Crawl4AI Documentation (v0.9.x)\nSection: Interactive Tutorial & Live Demo\n\nWant to learn by doing? We've got you covered:\n\n**🚀 [Live Demo](https://docs.crawl4ai.com/apps/c4a-script/)** - Try C4A-Script in your browser right now!\n\n**📁 [Tutorial…", "code_blocks": [{"language": "bash", "code": "# Clone and navigate to the tutorial\ncd docs/examples/c4a_script/tutorial/\n\n# Install dependencies\npip install -r requirements.txt\n\n# Launch the tutorial server\npython server.py\n\n# Open…", "filename": ""}], "chunk_position": 36, "heading_path": "Interactive Tutorial & Live Demo > Interactive Tutorial & Live Demo", "breadcrumbs": "C4A-Script - Crawl4AI Documentation (v0.9.x) > Interactive Tutorial & Live Demo > Interactive Tutorial & Live Demo"}, {"id": "418880369224ddf7", "url": "https://docs.crawl4ai.com/core/c4a-script/", "page_title": "C4A-Script - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "A comprehensive guide to C4A-Script, a human-readable DSL for web automation, covering syntax, commands, examples, and advanced features.", "heading": "Core Concepts", "content": "Page: C4A-Script - Crawl4AI Documentation (v0.9.x)\nSection: Core Concepts\n\n###### Commands and Syntax\n\nC4A-Script uses simple, English-like commands. Each command does one specific thing:\n\n###### Selectors: Finding Elements\n\nC4A-Script uses CSS selectors to identify elements on…", "code_blocks": [{"language": "text", "code": "# Comments start with #\nCOMMAND parameter1 parameter2\n\n# Most commands use CSS selectors in backticks\nCLICK `#submit-button`\n\n# Text content goes in quotes\nTYPE \"Hello, World!\"\n\n# Numbers are used…", "filename": ""}, {"language": "text", "code": "# By ID\nCLICK `#login-button`\n\n# By class\nCLICK `.submit-btn`\n\n# By attribute\nCLICK `button[type=\"submit\"]`\n\n# By accessible attributes\nCLICK `button[aria-label=\"Search\"][title=\"Search\"]`\n\n# Complex…", "filename": ""}, {"language": "text", "code": "# Set a variable\nSETVAR username = \"john@example.com\"\nSETVAR password = \"secret123\"\n\n# Use variables (prefix with $)\nTYPE $username\nPRESS Tab\nTYPE $password\nCopy", "filename": ""}], "chunk_position": 36, "heading_path": "Core Concepts > Core Concepts", "breadcrumbs": "C4A-Script - Crawl4AI Documentation (v0.9.x) > Core Concepts > Core Concepts"}, {"id": "30d5116394ae8feb", "url": "https://docs.crawl4ai.com/core/c4a-script/", "page_title": "C4A-Script - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "A comprehensive guide to C4A-Script, a human-readable DSL for web automation, covering syntax, commands, examples, and advanced features.", "heading": "Command Categories", "content": "Page: C4A-Script - Crawl4AI Documentation (v0.9.x)\nSection: Command Categories\n\n###### 🧭 Navigation Commands\n\nMove around the web like a user would:\n\n| Command | Purpose | Example |\n| --- | --- | --- |\n| `GO` | Navigate to URL | `GO https://example.com` |\n| `RELOAD` | Refresh…", "code_blocks": [], "chunk_position": 36, "heading_path": "Command Categories > Command Categories", "breadcrumbs": "C4A-Script - Crawl4AI Documentation (v0.9.x) > Command Categories > Command Categories"}, {"id": "1b87e28412525e33", "url": "https://docs.crawl4ai.com/core/c4a-script/", "page_title": "C4A-Script - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "A comprehensive guide to C4A-Script, a human-readable DSL for web automation, covering syntax, commands, examples, and advanced features.", "heading": "Real-World Examples", "content": "Page: C4A-Script - Crawl4AI Documentation (v0.9.x)\nSection: Real-World Examples\n\n###### Example 1: Login Flow\n\n###### Example 2: E-commerce Shopping\n\n###### Example 3: Form Automation with Conditions", "code_blocks": [{"language": "text", "code": "# Complete login automation\nGO https://myapp.com/login\n\n# Wait for page to load\nWAIT `#login-form` 5\n\n# Fill credentials\nCLICK `#email`\nTYPE \"user@example.com\"\nPRESS Tab\nTYPE \"mypassword\"\n\n# Submit…", "filename": ""}, {"language": "text", "code": "# Shopping automation with variables\nSETVAR product = \"laptop\"\nSETVAR budget = \"1000\"\n\nGO https://shop.example.com\nWAIT `#search-box` 3\n\n# Search for product\nTYPE $product\nPRESS Enter\nWAIT…", "filename": ""}, {"language": "text", "code": "# Smart form filling with error handling\nGO https://forms.example.com\n\n# Check if user is already logged in\nIF (EXISTS `.user-menu`) THEN GO https://forms.example.com/new\nIF (NOT EXISTS `.user-menu`)…", "filename": ""}], "chunk_position": 36, "heading_path": "Real-World Examples > Real-World Examples", "breadcrumbs": "C4A-Script - Crawl4AI Documentation (v0.9.x) > Real-World Examples > Real-World Examples"}, {"id": "3378c8234cd05731", "url": "https://docs.crawl4ai.com/core/c4a-script/", "page_title": "C4A-Script - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "A comprehensive guide to C4A-Script, a human-readable DSL for web automation, covering syntax, commands, examples, and advanced features.", "heading": "Visual Programming with Blockly", "content": "Page: C4A-Script - Crawl4AI Documentation (v0.9.x)\nSection: Visual Programming with Blockly\n\nC4A-Script includes a powerful visual programming interface built on Google Blockly. Perfect for:\n\n- **Non-programmers** who want to create automation\n- **Rapid prototyping** of automation…", "code_blocks": [], "chunk_position": 36, "heading_path": "Visual Programming with Blockly > Visual Programming with Blockly", "breadcrumbs": "C4A-Script - Crawl4AI Documentation (v0.9.x) > Visual Programming with Blockly > Visual Programming with Blockly"}, {"id": "1d546e4839cf3032", "url": "https://docs.crawl4ai.com/core/c4a-script/", "page_title": "C4A-Script - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "A comprehensive guide to C4A-Script, a human-readable DSL for web automation, covering syntax, commands, examples, and advanced features.", "heading": "Advanced Features", "content": "Page: C4A-Script - Crawl4AI Documentation (v0.9.x)\nSection: Advanced Features\n\n###### Recording Mode\n\nThe tutorial interface includes a recording feature that watches your browser interactions and automatically generates C4A-Script commands:\n\n- Click \"Record\" in the tutorial…", "code_blocks": [{"language": "text", "code": "# Use comments for debugging\n# This will wait up to 10 seconds for the element\nWAIT `#slow-loading-element` 10\n\n# Check if element exists before clicking\nIF (EXISTS `#optional-button`) THEN CLICK…", "filename": ""}, {"language": "python", "code": "from crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\n# Use C4A-Script for interaction before crawling\nscript = \"\"\"\nGO https://example.com\nCLICK `#load-more-content`\nWAIT `.dynamic-content`…", "filename": ""}], "chunk_position": 36, "heading_path": "Advanced Features > Advanced Features", "breadcrumbs": "C4A-Script - Crawl4AI Documentation (v0.9.x) > Advanced Features > Advanced Features"}, {"id": "9951b8b64d8663d7", "url": "https://docs.crawl4ai.com/core/c4a-script/", "page_title": "C4A-Script - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "A comprehensive guide to C4A-Script, a human-readable DSL for web automation, covering syntax, commands, examples, and advanced features.", "heading": "Best Practices", "content": "Page: C4A-Script - Crawl4AI Documentation (v0.9.x)\nSection: Best Practices\n\n###### 1. Always Wait for Elements\n\n###### 2. Use Descriptive Comments\n\n###### 3. Handle Variable Conditions\n\n###### 4. Use Variables for Reusability", "code_blocks": [{"language": "text", "code": "# Bad: Clicking immediately\nCLICK `#button`\n\n# Good: Wait for element to appear\nWAIT `#button` 5\nCLICK `#button`\nCopy", "filename": ""}, {"language": "text", "code": "# Login to user account\nGO https://myapp.com/login\nWAIT `#login-form` 5\n\n# Enter credentials\nTYPE \"user@example.com\"\nPRESS Tab\nTYPE \"password123\"\n\n# Submit and wait for redirect\nCLICK…", "filename": ""}, {"language": "text", "code": "# Handle different page states\nIF (EXISTS `.cookie-banner`) THEN CLICK `.accept-cookies`\nIF (EXISTS `.popup-modal`) THEN CLICK `.close-modal`\n\n# Proceed with main workflow\nCLICK `#main-action`\nCopy", "filename": ""}, {"language": "text", "code": "# Define once, use everywhere\nSETVAR base_url = \"https://myapp.com\"\nSETVAR test_email = \"test@example.com\"\n\nGO $base_url/login\nSET `#email` $test_email\nCopy", "filename": ""}], "chunk_position": 36, "heading_path": "Best Practices > Best Practices", "breadcrumbs": "C4A-Script - Crawl4AI Documentation (v0.9.x) > Best Practices > Best Practices"}, {"id": "9d344d3c9d276d76", "url": "https://docs.crawl4ai.com/core/c4a-script/", "page_title": "C4A-Script - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "A comprehensive guide to C4A-Script, a human-readable DSL for web automation, covering syntax, commands, examples, and advanced features.", "heading": "Getting Help", "content": "Page: C4A-Script - Crawl4AI Documentation (v0.9.x)\nSection: Getting Help\n\n- **📖 [Complete Examples](/examples/c4a_script/)** - Real-world automation scripts\n- **🎮 [Interactive Tutorial](/examples/c4a_script/tutorial/)** - Hands-on learning environment\n- **📋 [API…", "code_blocks": [], "chunk_position": 36, "heading_path": "Getting Help > Getting Help", "breadcrumbs": "C4A-Script - Crawl4AI Documentation (v0.9.x) > Getting Help > Getting Help"}, {"id": "f93777d4284b3e73", "url": "https://docs.crawl4ai.com/core/c4a-script/", "page_title": "C4A-Script - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "A comprehensive guide to C4A-Script, a human-readable DSL for web automation, covering syntax, commands, examples, and advanced features.", "heading": "What's Next?", "content": "Page: C4A-Script - Crawl4AI Documentation (v0.9.x)\nSection: What's Next?\n\nReady to dive deeper? Check out:\n\n- **[API Reference](/api/c4a-script-reference/)** - Complete command documentation\n- **[Tutorial Examples](/examples/c4a_script/)** - Copy-paste ready scripts\n-…", "code_blocks": [], "chunk_position": 36, "heading_path": "What's Next? > What's Next?", "breadcrumbs": "C4A-Script - Crawl4AI Documentation (v0.9.x) > What's Next? > What's Next?"}, {"id": "bb546b1fe020ca0a", "url": "https://docs.crawl4ai.com/core/cache-modes/", "page_title": "Crawl4AI Cache System and Migration Guide", "page_type": "guide", "page_summary": "This page explains the new CacheMode enum introduced in Crawl4AI v0.5.0, which replaces old boolean cache flags, and provides migration examples and a mapping table for transitioning from legacy…", "heading": "Overview", "content": "Page: Crawl4AI Cache System and Migration Guide\nSection: Overview\n\nStarting from version 0.5.0, Crawl4AI introduces a new caching system that replaces the old boolean flags with a more intuitive `CacheMode` enum. This change simplifies cache control and makes the…", "code_blocks": [], "chunk_position": 37, "heading_path": "Overview > Overview", "breadcrumbs": "Crawl4AI Cache System and Migration Guide > Overview > Overview"}, {"id": "692be31a34a25a5e", "url": "https://docs.crawl4ai.com/core/cache-modes/", "page_title": "Crawl4AI Cache System and Migration Guide", "page_type": "guide", "page_summary": "This page explains the new CacheMode enum introduced in Crawl4AI v0.5.0, which replaces old boolean cache flags, and provides migration examples and a mapping table for transitioning from legacy…", "heading": "Old vs New Approach", "content": "Page: Crawl4AI Cache System and Migration Guide\nSection: Old vs New Approach\n\nThe old system used multiple boolean flags:\n- `bypass_cache`: Skip cache entirely\n- `disable_cache`: Disable all caching\n- `no_cache_read`: Don't read from cache\n- `no_cache_write`: Don't write to…", "code_blocks": [], "chunk_position": 37, "heading_path": "Old vs New Approach > Old vs New Approach", "breadcrumbs": "Crawl4AI Cache System and Migration Guide > Old vs New Approach > Old vs New Approach"}, {"id": "995827632f38f2fd", "url": "https://docs.crawl4ai.com/core/cache-modes/", "page_title": "Crawl4AI Cache System and Migration Guide", "page_type": "guide", "page_summary": "This page explains the new CacheMode enum introduced in Crawl4AI v0.5.0, which replaces old boolean cache flags, and provides migration examples and a mapping table for transitioning from legacy…", "heading": "Common Migration Patterns", "content": "Page: Crawl4AI Cache System and Migration Guide\nSection: Common Migration Patterns\n\n| Legacy Flag | Replacement |\n| --- | --- |\n| `bypass_cache` | `cache_mode=CacheMode.BYPASS` |\n| `disable_cache` | `cache_mode=CacheMode.DISABLED` |\n| `no_cache_read` |…", "code_blocks": [], "chunk_position": 37, "heading_path": "Common Migration Patterns > Common Migration Patterns", "breadcrumbs": "Crawl4AI Cache System and Migration Guide > Common Migration Patterns > Common Migration Patterns"}, {"id": "970e395a7a1a50ac", "url": "https://docs.crawl4ai.com/core/cli/", "page_title": "Command Line Interface - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Crawl4AI CLI (`crwl`), covering installation, basic usage, configuration options (browser, crawler, extraction), advanced features like LLM Q&A…", "heading": "Installation", "content": "Page: Command Line Interface - Crawl4AI Documentation (v0.9.x)\nSection: Installation\n\nThe Crawl4AI CLI will be installed automatically when you install the library.", "code_blocks": [], "chunk_position": 38, "heading_path": "Installation > Installation", "breadcrumbs": "Command Line Interface - Crawl4AI Documentation (v0.9.x) > Installation > Installation"}, {"id": "2a79d0cdbfeaa9e7", "url": "https://docs.crawl4ai.com/core/cli/", "page_title": "Command Line Interface - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Crawl4AI CLI (`crwl`), covering installation, basic usage, configuration options (browser, crawler, extraction), advanced features like LLM Q&A…", "heading": "Basic Usage", "content": "Page: Command Line Interface - Crawl4AI Documentation (v0.9.x)\nSection: Basic Usage\n\nThe Crawl4AI CLI (`crwl`) provides a simple interface to the Crawl4AI library:", "code_blocks": [{"language": "bash", "code": "# Basic crawling\ncrwl https://example.com\n\n# Get markdown output\ncrwl https://example.com -o markdown\n\n# Verbose JSON output with cache bypass\ncrwl https://example.com -o json -v --bypass-cache\n\n#…", "filename": ""}], "chunk_position": 38, "heading_path": "Basic Usage > Basic Usage", "breadcrumbs": "Command Line Interface - Crawl4AI Documentation (v0.9.x) > Basic Usage > Basic Usage"}, {"id": "29e8760b7a66e345", "url": "https://docs.crawl4ai.com/core/cli/", "page_title": "Command Line Interface - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Crawl4AI CLI (`crwl`), covering installation, basic usage, configuration options (browser, crawler, extraction), advanced features like LLM Q&A…", "heading": "Quick Example of Advanced Usage", "content": "Page: Command Line Interface - Crawl4AI Documentation (v0.9.x)\nSection: Quick Example of Advanced Usage\n\nIf you clone the repository and run the following command, you will receive the content of the page in JSON format according to a JSON-CSS schema:", "code_blocks": [{"language": "bash", "code": "crwl \"https://www.infoq.com/ai-ml-data-eng/\" -e docs/examples/cli/extract_css.yml -s docs/examples/cli/css_schema.json -o json;\nCopy", "filename": ""}], "chunk_position": 38, "heading_path": "Quick Example of Advanced Usage > Quick Example of Advanced Usage", "breadcrumbs": "Command Line Interface - Crawl4AI Documentation (v0.9.x) > Quick Example of Advanced Usage > Quick Example of Advanced Usage"}, {"id": "b297b1db400623b9", "url": "https://docs.crawl4ai.com/core/cli/", "page_title": "Command Line Interface - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Crawl4AI CLI (`crwl`), covering installation, basic usage, configuration options (browser, crawler, extraction), advanced features like LLM Q&A…", "heading": "Browser Configuration", "content": "Page: Command Line Interface - Crawl4AI Documentation (v0.9.x)\nSection: Browser Configuration\n\nBrowser settings can be configured via YAML file or command line parameters:", "code_blocks": [{"language": "yaml", "code": "# browser.yml\nheadless: true\nviewport_width: 1280\nuser_agent_mode: \"random\"\nverbose: true\nignore_https_errors: true\nCopy", "filename": "browser.yml"}, {"language": "bash", "code": "# Using config file\ncrwl https://example.com -B browser.yml\n\n# Using direct parameters\ncrwl https://example.com -b \"headless=true,viewport_width=1280,user_agent_mode=random\"\nCopy", "filename": ""}], "chunk_position": 38, "heading_path": "Browser Configuration > Browser Configuration", "breadcrumbs": "Command Line Interface - Crawl4AI Documentation (v0.9.x) > Browser Configuration > Browser Configuration"}, {"id": "a727a733893d1432", "url": "https://docs.crawl4ai.com/core/cli/", "page_title": "Command Line Interface - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Crawl4AI CLI (`crwl`), covering installation, basic usage, configuration options (browser, crawler, extraction), advanced features like LLM Q&A…", "heading": "Extraction Configuration", "content": "Page: Command Line Interface - Crawl4AI Documentation (v0.9.x)\nSection: Extraction Configuration\n\nTwo types of extraction are supported:\n\n- CSS/XPath-based extraction:", "code_blocks": [{"language": "yaml", "code": "# extract_css.yml\ntype: \"json-css\"\nparams:\n  verbose: true\nCopy", "filename": "extract_css.yml"}, {"language": "json", "code": "// css_schema.json\n{\n  \"name\": \"ArticleExtractor\",\n  \"baseSelector\": \".article\",\n  \"fields\": [\n    {\n      \"name\": \"title\",\n      \"selector\": \"h1.title\",\n      \"type\": \"text\"\n    },\n    {…", "filename": "css_schema.json"}, {"language": "yaml", "code": "# extract_llm.yml\ntype: \"llm\"\nprovider: \"openai/gpt-4\"\ninstruction: \"Extract all articles with their titles and links\"\napi_token: \"your-token\"\nparams:\n  temperature: 0.3\n  max_tokens: 1000\nCopy", "filename": "extract_llm.yml"}, {"language": "json", "code": "// llm_schema.json\n{\n  \"title\": \"Article\",\n  \"type\": \"object\",\n  \"properties\": {\n    \"title\": {\n      \"type\": \"string\",\n      \"description\": \"The title of the article\"\n    },\n    \"link\": {…", "filename": "llm_schema.json"}], "chunk_position": 38, "heading_path": "Extraction Configuration > Extraction Configuration", "breadcrumbs": "Command Line Interface - Crawl4AI Documentation (v0.9.x) > Extraction Configuration > Extraction Configuration"}, {"id": "f5744a86ebc4a0cd", "url": "https://docs.crawl4ai.com/core/cli/", "page_title": "Command Line Interface - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Crawl4AI CLI (`crwl`), covering installation, basic usage, configuration options (browser, crawler, extraction), advanced features like LLM Q&A…", "heading": "LLM Q&A", "content": "Page: Command Line Interface - Crawl4AI Documentation (v0.9.x)\nSection: LLM Q&A\n\nAsk questions about crawled content:\n\nFirst-time setup:\n- Prompts for LLM provider and API token\n- Saves configuration in `~/.crawl4ai/global.yml`\n- Supports various providers (openai/gpt-4,…", "code_blocks": [{"language": "bash", "code": "# Simple question\ncrwl https://example.com -q \"What is the main topic discussed?\"\n\n# View content then ask questions\ncrwl https://example.com -o markdown  # See content first\ncrwl https://example.com…", "filename": ""}], "chunk_position": 38, "heading_path": "LLM Q&A > LLM Q&A", "breadcrumbs": "Command Line Interface - Crawl4AI Documentation (v0.9.x) > LLM Q&A > LLM Q&A"}, {"id": "6ca7f09925049b86", "url": "https://docs.crawl4ai.com/core/cli/", "page_title": "Command Line Interface - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Crawl4AI CLI (`crwl`), covering installation, basic usage, configuration options (browser, crawler, extraction), advanced features like LLM Q&A…", "heading": "Structured Data Extraction", "content": "Page: Command Line Interface - Crawl4AI Documentation (v0.9.x)\nSection: Structured Data Extraction\n\nExtract structured data using CSS selectors:\n\nOr using LLM-based extraction:", "code_blocks": [{"language": "bash", "code": "crwl https://example.com \\\n    -e extract_css.yml \\\n    -s css_schema.json \\\n    -o json\nCopy", "filename": ""}, {"language": "bash", "code": "crwl https://example.com \\\n    -e extract_llm.yml \\\n    -s llm_schema.json \\\n    -o json\nCopy", "filename": ""}], "chunk_position": 38, "heading_path": "Structured Data Extraction > Structured Data Extraction", "breadcrumbs": "Command Line Interface - Crawl4AI Documentation (v0.9.x) > Structured Data Extraction > Structured Data Extraction"}, {"id": "d093eef0ec5eb79e", "url": "https://docs.crawl4ai.com/core/cli/", "page_title": "Command Line Interface - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Crawl4AI CLI (`crwl`), covering installation, basic usage, configuration options (browser, crawler, extraction), advanced features like LLM Q&A…", "heading": "Output Formats", "content": "Page: Command Line Interface - Crawl4AI Documentation (v0.9.x)\nSection: Output Formats\n\n- `all` - Full crawl result including metadata\n- `json` - Extracted structured data (when using extraction)\n- `markdown` / `md` - Raw markdown output\n- `markdown-fit` / `md-fit` - Filtered markdown…", "code_blocks": [], "chunk_position": 38, "heading_path": "Output Formats > Output Formats", "breadcrumbs": "Command Line Interface - Crawl4AI Documentation (v0.9.x) > Output Formats > Output Formats"}, {"id": "8f5cb5c6a59cea92", "url": "https://docs.crawl4ai.com/core/cli/", "page_title": "Command Line Interface - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Crawl4AI CLI (`crwl`), covering installation, basic usage, configuration options (browser, crawler, extraction), advanced features like LLM Q&A…", "heading": "Complete Examples", "content": "Page: Command Line Interface - Crawl4AI Documentation (v0.9.x)\nSection: Complete Examples\n\n- Basic Extraction:\n- Structured Data Extraction:\n- LLM Extraction with Filtering:\n- Interactive Q&A:", "code_blocks": [{"language": "bash", "code": "crwl https://example.com \\\n    -B browser.yml \\\n    -C crawler.yml \\\n    -o json\nCopy", "filename": ""}, {"language": "bash", "code": "crwl https://example.com \\\n    -e extract_css.yml \\\n    -s css_schema.json \\\n    -o json \\\n    -v\nCopy", "filename": ""}, {"language": "bash", "code": "crwl https://example.com \\\n    -B browser.yml \\\n    -e extract_llm.yml \\\n    -s llm_schema.json \\\n    -f filter_bm25.yml \\\n    -o json\nCopy", "filename": ""}, {"language": "bash", "code": "# First crawl and view\ncrwl https://example.com -o markdown\n\n# Then ask questions\ncrwl https://example.com -q \"What are the main points?\"\ncrwl https://example.com -q \"Summarize the conclusions\"\nCopy", "filename": ""}], "chunk_position": 38, "heading_path": "Complete Examples > Complete Examples", "breadcrumbs": "Command Line Interface - Crawl4AI Documentation (v0.9.x) > Complete Examples > Complete Examples"}, {"id": "daa4740c1837153c", "url": "https://docs.crawl4ai.com/core/cli/", "page_title": "Command Line Interface - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Crawl4AI CLI (`crwl`), covering installation, basic usage, configuration options (browser, crawler, extraction), advanced features like LLM Q&A…", "heading": "Best Practices & Tips", "content": "Page: Command Line Interface - Crawl4AI Documentation (v0.9.x)\nSection: Best Practices & Tips\n\n- **Configuration Management** :\n  - Keep common configurations in YAML files\n  - Use CLI parameters for quick overrides\n  - Store sensitive data (API tokens) in `~/.crawl4ai/global.yml`\n-…", "code_blocks": [], "chunk_position": 38, "heading_path": "Best Practices & Tips > Best Practices & Tips", "breadcrumbs": "Command Line Interface - Crawl4AI Documentation (v0.9.x) > Best Practices & Tips > Best Practices & Tips"}, {"id": "88ee84bc14a290b1", "url": "https://docs.crawl4ai.com/core/cli/", "page_title": "Command Line Interface - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Crawl4AI CLI (`crwl`), covering installation, basic usage, configuration options (browser, crawler, extraction), advanced features like LLM Q&A…", "heading": "Recap", "content": "Page: Command Line Interface - Crawl4AI Documentation (v0.9.x)\nSection: Recap\n\nThe Crawl4AI CLI provides:\n- Flexible configuration via files and parameters\n- Multiple extraction strategies (CSS, XPath, LLM)\n- Content filtering and optimization\n- Interactive Q&A capabilities\n-…", "code_blocks": [], "chunk_position": 38, "heading_path": "Recap > Recap", "breadcrumbs": "Command Line Interface - Crawl4AI Documentation (v0.9.x) > Recap > Recap"}, {"id": "bf6b740b99b3e59b", "url": "https://docs.crawl4ai.com/core/content-selection/", "page_title": "Content Selection", "page_type": "api", "page_summary": "This page explains how to select, filter, and refine content from crawls using CrawlerRunConfig parameters, including CSS selectors, content filtering, iframe handling, shadow DOM flattening, and…", "heading": "1. CSS-Based Selection", "content": "Page: Content Selection\nSection: 1. CSS-Based Selection\n\nCrawl4AI provides multiple ways to select, filter, and refine the content from your crawls. Whether you need to target a specific CSS region, exclude entire tags, filter out external links, or remove…", "code_blocks": [], "chunk_position": 39, "heading_path": "1. CSS-Based Selection > 1. CSS-Based Selection", "breadcrumbs": "Content Selection > 1. CSS-Based Selection > 1. CSS-Based Selection"}, {"id": "29bd71f72946bdc8", "url": "https://docs.crawl4ai.com/core/content-selection/", "page_title": "Content Selection", "page_type": "api", "page_summary": "This page explains how to select, filter, and refine content from crawls using CrawlerRunConfig parameters, including CSS selectors, content filtering, iframe handling, shadow DOM flattening, and…", "heading": "1.1 Using css_selector", "content": "Page: Content Selection\nSection: 1.1 Using css_selector\n\nA straightforward way to limit your crawl results to a certain region of the page is css_selector in CrawlerRunConfig:\n\nResult: Only elements matching that selector remain in result.cleaned_html.", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nasync def main():\n    config = CrawlerRunConfig(\n        # e.g., first 30 items from Hacker News…", "filename": ""}], "chunk_position": 39, "heading_path": "1.1 Using css_selector > 1.1 Using css_selector", "breadcrumbs": "Content Selection > 1.1 Using css_selector > 1.1 Using css_selector"}, {"id": "65e9fa33535c191a", "url": "https://docs.crawl4ai.com/core/content-selection/", "page_title": "Content Selection", "page_type": "api", "page_summary": "This page explains how to select, filter, and refine content from crawls using CrawlerRunConfig parameters, including CSS selectors, content filtering, iframe handling, shadow DOM flattening, and…", "heading": "1.2 Using target_elements", "content": "Page: Content Selection\nSection: 1.2 Using target_elements\n\nThe target_elements parameter provides more flexibility by allowing you to target multiple elements for content extraction while preserving the entire page context for other features:\n\nKey…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nasync def main():\n    config = CrawlerRunConfig(\n        # Target article body and sidebar, but not other content…", "filename": ""}], "chunk_position": 39, "heading_path": "1.2 Using target_elements > 1.2 Using target_elements", "breadcrumbs": "Content Selection > 1.2 Using target_elements > 1.2 Using target_elements"}, {"id": "64caf6a96ade425f", "url": "https://docs.crawl4ai.com/core/content-selection/", "page_title": "Content Selection", "page_type": "api", "page_summary": "This page explains how to select, filter, and refine content from crawls using CrawlerRunConfig parameters, including CSS selectors, content filtering, iframe handling, shadow DOM flattening, and…", "heading": "2.1 Basic Overview", "content": "Page: Content Selection\nSection: 2.1 Basic Overview\n\nExplanation:\n\n- word_count_threshold: Ignores text blocks under X words. Helps skip trivial blocks like short nav or disclaimers.\n- excluded_tags: Removes entire tags (<form>, , <footer>,…", "code_blocks": [{"language": "python", "code": "config = CrawlerRunConfig(\n    # Content thresholds\n    word_count_threshold=10,        # Minimum words per block\n\n    # Tag exclusions\n    excluded_tags=['form', 'header', 'footer', 'nav'],\n\n    #…", "filename": ""}, {"language": "python", "code": "[\n    'facebook.com',\n    'twitter.com',\n    'x.com',\n    'linkedin.com',\n    'instagram.com',\n    'pinterest.com',\n    'tiktok.com',\n    'snapchat.com',\n    'reddit.com',\n]", "filename": ""}], "chunk_position": 39, "heading_path": "2.1 Basic Overview > 2.1 Basic Overview", "breadcrumbs": "Content Selection > 2.1 Basic Overview > 2.1 Basic Overview"}, {"id": "465bc262ca784499", "url": "https://docs.crawl4ai.com/core/content-selection/", "page_title": "Content Selection", "page_type": "api", "page_summary": "This page explains how to select, filter, and refine content from crawls using CrawlerRunConfig parameters, including CSS selectors, content filtering, iframe handling, shadow DOM flattening, and…", "heading": "2.2 Example Usage", "content": "Page: Content Selection\nSection: 2.2 Example Usage\n\nNote: If these parameters remove too much, reduce or disable them accordingly.", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\n\nasync def main():\n    config = CrawlerRunConfig(\n        css_selector=\"main.content\",…", "filename": ""}], "chunk_position": 39, "heading_path": "2.2 Example Usage > 2.2 Example Usage", "breadcrumbs": "Content Selection > 2.2 Example Usage > 2.2 Example Usage"}, {"id": "6c114310b4dddf52", "url": "https://docs.crawl4ai.com/core/content-selection/", "page_title": "Content Selection", "page_type": "api", "page_summary": "This page explains how to select, filter, and refine content from crawls using CrawlerRunConfig parameters, including CSS selectors, content filtering, iframe handling, shadow DOM flattening, and…", "heading": "3. Handling Iframes", "content": "Page: Content Selection\nSection: 3. Handling Iframes\n\nSome sites embed content in <iframe> tags. If you want that inline:\n\nUsage:", "code_blocks": [{"language": "python", "code": "config = CrawlerRunConfig(\n    # Merge iframe content into the final output\n    process_iframes=True,\n    remove_overlay_elements=True,\n    # Remove GDPR/cookie consent popups (OneTrust, Cookiebot,…", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nasync def main():\n    config = CrawlerRunConfig(\n        process_iframes=True,\n        remove_overlay_elements=True\n    )…", "filename": ""}], "chunk_position": 39, "heading_path": "3. Handling Iframes > 3. Handling Iframes", "breadcrumbs": "Content Selection > 3. Handling Iframes > 3. Handling Iframes"}, {"id": "449854c02c6bcc60", "url": "https://docs.crawl4ai.com/core/content-selection/", "page_title": "Content Selection", "page_type": "api", "page_summary": "This page explains how to select, filter, and refine content from crawls using CrawlerRunConfig parameters, including CSS selectors, content filtering, iframe handling, shadow DOM flattening, and…", "heading": "3.1 Flattening Shadow DOM", "content": "Page: Content Selection\nSection: 3.1 Flattening Shadow DOM\n\nSites built with Web Components (Stencil, Lit, Shoelace, Angular Elements, etc.) render content inside Shadow DOM — an encapsulated sub-tree that is invisible to normal page serialization. The…", "code_blocks": [{"language": "python", "code": "config = CrawlerRunConfig(\n    # Flatten shadow DOM into the main document\n    flatten_shadow_dom=True,\n    # Give web components time to hydrate\n    wait_until=\"load\",…", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nasync def main():\n    config = CrawlerRunConfig(\n        flatten_shadow_dom=True,\n        wait_until=\"load\",…", "filename": ""}], "chunk_position": 39, "heading_path": "3.1 Flattening Shadow DOM > 3.1 Flattening Shadow DOM", "breadcrumbs": "Content Selection > 3.1 Flattening Shadow DOM > 3.1 Flattening Shadow DOM"}, {"id": "5c4481646f6ec0c1", "url": "https://docs.crawl4ai.com/core/content-selection/", "page_title": "Content Selection", "page_type": "api", "page_summary": "This page explains how to select, filter, and refine content from crawls using CrawlerRunConfig parameters, including CSS selectors, content filtering, iframe handling, shadow DOM flattening, and…", "heading": "4. Structured Extraction Examples", "content": "Page: Content Selection\nSection: 4. Structured Extraction Examples\n\nYou can combine content selection with a more advanced extraction strategy. For instance, a CSS-based or LLM-based extraction strategy can run on the filtered HTML.", "code_blocks": [], "chunk_position": 39, "heading_path": "4. Structured Extraction Examples > 4. Structured Extraction Examples", "breadcrumbs": "Content Selection > 4. Structured Extraction Examples > 4. Structured Extraction Examples"}, {"id": "b52f68cfc5f10eab", "url": "https://docs.crawl4ai.com/core/content-selection/", "page_title": "Content Selection", "page_type": "api", "page_summary": "This page explains how to select, filter, and refine content from crawls using CrawlerRunConfig parameters, including CSS selectors, content filtering, iframe handling, shadow DOM flattening, and…", "heading": "4.1 Pattern-Based with JsonCssExtractionStrategy", "content": "Page: Content Selection\nSection: 4.1 Pattern-Based with JsonCssExtractionStrategy\n\n```python\nimport asyncio\nimport json\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\nfrom crawl4ai import JsonCssExtractionStrategy\n\nasync def main():\n    # Minimal schema for repeated items…\n```", "code_blocks": [{"language": "python", "code": "import asyncio\nimport json\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\nfrom crawl4ai import JsonCssExtractionStrategy\n\nasync def main():\n    # Minimal schema for repeated items…", "filename": ""}], "chunk_position": 39, "heading_path": "4.1 Pattern-Based with JsonCssExtractionStrategy > 4.1 Pattern-Based with JsonCssExtractionStrategy", "breadcrumbs": "Content Selection > 4.1 Pattern-Based with JsonCssExtractionStrategy > 4.1 Pattern-Based with JsonCssExtractionStrategy"}, {"id": "398a6d3a6e839430", "url": "https://docs.crawl4ai.com/core/content-selection/", "page_title": "Content Selection", "page_type": "api", "page_summary": "This page explains how to select, filter, and refine content from crawls using CrawlerRunConfig parameters, including CSS selectors, content filtering, iframe handling, shadow DOM flattening, and…", "heading": "4.2 LLM-Based Extraction", "content": "Page: Content Selection\nSection: 4.2 LLM-Based Extraction\n\nHere, the crawler:\n\n- Filters out external links (exclude_external_links=True).\n- Ignores very short text blocks (word_count_threshold=20).\n- Passes the final HTML to your LLM strategy for an…", "code_blocks": [{"language": "python", "code": "import asyncio\nimport json\nfrom pydantic import BaseModel, Field\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, LLMConfig\nfrom crawl4ai import LLMExtractionStrategy\n\nclass…", "filename": ""}], "chunk_position": 39, "heading_path": "4.2 LLM-Based Extraction > 4.2 LLM-Based Extraction", "breadcrumbs": "Content Selection > 4.2 LLM-Based Extraction > 4.2 LLM-Based Extraction"}, {"id": "14c839e4c8012a70", "url": "https://docs.crawl4ai.com/core/content-selection/", "page_title": "Content Selection", "page_type": "api", "page_summary": "This page explains how to select, filter, and refine content from crawls using CrawlerRunConfig parameters, including CSS selectors, content filtering, iframe handling, shadow DOM flattening, and…", "heading": "5. Comprehensive Example", "content": "Page: Content Selection\nSection: 5. Comprehensive Example\n\nBelow is a short function that unifies CSS selection, exclusion logic, and a pattern-based extraction, demonstrating how you can fine-tune your final data:\n\nWhy This Works:\n- CSS scoping with…", "code_blocks": [{"language": "python", "code": "import asyncio\nimport json\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\nfrom crawl4ai import JsonCssExtractionStrategy\n\nasync def extract_main_articles(url: str):\n    schema = {…", "filename": ""}], "chunk_position": 39, "heading_path": "5. Comprehensive Example > 5. Comprehensive Example", "breadcrumbs": "Content Selection > 5. Comprehensive Example > 5. Comprehensive Example"}, {"id": "ac3f7a2db966d250", "url": "https://docs.crawl4ai.com/core/content-selection/", "page_title": "Content Selection", "page_type": "api", "page_summary": "This page explains how to select, filter, and refine content from crawls using CrawlerRunConfig parameters, including CSS selectors, content filtering, iframe handling, shadow DOM flattening, and…", "heading": "6. Scraping Modes", "content": "Page: Content Selection\nSection: 6. Scraping Modes\n\nCrawl4AI uses LXMLWebScrapingStrategy (LXML-based) as the default scraping strategy for HTML content processing. This strategy offers excellent performance, especially for large HTML…", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler, CrawlerRunConfig, LXMLWebScrapingStrategy\n\nasync def main():\n    # Default configuration already uses LXMLWebScrapingStrategy\n    config = CrawlerRunConfig()…", "filename": ""}, {"language": "python", "code": "from crawl4ai import ContentScrapingStrategy, ScrapingResult, MediaItem, Media, Link, Links\n\nclass CustomScrapingStrategy(ContentScrapingStrategy):\n    def scrap(self, url: str, html: str, **kwargs)…", "filename": ""}], "chunk_position": 39, "heading_path": "6. Scraping Modes > 6. Scraping Modes", "breadcrumbs": "Content Selection > 6. Scraping Modes > 6. Scraping Modes"}, {"id": "5072b58caf35bfe4", "url": "https://docs.crawl4ai.com/core/content-selection/", "page_title": "Content Selection", "page_type": "api", "page_summary": "This page explains how to select, filter, and refine content from crawls using CrawlerRunConfig parameters, including CSS selectors, content filtering, iframe handling, shadow DOM flattening, and…", "heading": "7. Combining CSS Selection Methods", "content": "Page: Content Selection\nSection: 7. Combining CSS Selection Methods\n\nYou can combine css_selector and target_elements in powerful ways to achieve fine-grained control over your output:\n\nThis approach gives you the best of both worlds:\n- Markdown generation and content…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\n\nasync def main():\n    # Target specific content but preserve page context\n    config = CrawlerRunConfig(\n        #…", "filename": ""}], "chunk_position": 39, "heading_path": "7. Combining CSS Selection Methods > 7. Combining CSS Selection Methods", "breadcrumbs": "Content Selection > 7. Combining CSS Selection Methods > 7. Combining CSS Selection Methods"}, {"id": "fc99e8e3228ef585", "url": "https://docs.crawl4ai.com/core/content-selection/", "page_title": "Content Selection", "page_type": "api", "page_summary": "This page explains how to select, filter, and refine content from crawls using CrawlerRunConfig parameters, including CSS selectors, content filtering, iframe handling, shadow DOM flattening, and…", "heading": "8. Conclusion", "content": "Page: Content Selection\nSection: 8. Conclusion\n\nBy mixing target_elements or css_selector scoping, content filtering parameters, and advanced extraction strategies, you can precisely choose which data to keep. Key parameters in CrawlerRunConfig…", "code_blocks": [], "chunk_position": 39, "heading_path": "8. Conclusion > 8. Conclusion", "breadcrumbs": "Content Selection > 8. Conclusion > 8. Conclusion"}, {"id": "47f62967a607117f", "url": "https://docs.crawl4ai.com/core/crawler-result/", "page_title": "Crawl Result and Output", "page_type": "api", "page_summary": "Describes the CrawlResult object returned by Crawl4AI's arun() method, including all fields, markdown generation, structured extraction, and additional outputs like links, media, tables, screenshots,…", "heading": "1. The `CrawlResult` Model", "content": "Page: Crawl Result and Output\nSection: 1. The `CrawlResult` Model\n\nBelow is the core schema. Each field captures a different aspect of the crawl’s result:\n\n```\nclass MarkdownGenerationResult(BaseModel):\n    raw_markdown: str\n    markdown_with_citations: str…", "code_blocks": [{"language": "python", "code": "class MarkdownGenerationResult(BaseModel):\n    raw_markdown: str\n    markdown_with_citations: str\n    references_markdown: str\n    fit_markdown: Optional[str] = None\n    fit_html: Optional[str] =…", "filename": ""}], "chunk_position": 40, "heading_path": "1. The `CrawlResult` Model > 1. The `CrawlResult` Model", "breadcrumbs": "Crawl Result and Output > 1. The `CrawlResult` Model > 1. The `CrawlResult` Model"}, {"id": "97b85e027d17b18a", "url": "https://docs.crawl4ai.com/core/crawler-result/", "page_title": "Crawl Result and Output", "page_type": "api", "page_summary": "Describes the CrawlResult object returned by Crawl4AI's arun() method, including all fields, markdown generation, structured extraction, and additional outputs like links, media, tables, screenshots,…", "heading": "2. HTML Variants", "content": "Page: Crawl Result and Output\nSection: 2. HTML Variants\n\n###### `html`: Raw HTML\n\nCrawl4AI preserves the exact HTML as `result.html`. Useful for:\n\n- Debugging page issues or checking the original content.\n- Performing your own specialized parse if…", "code_blocks": [{"language": "python", "code": "config = CrawlerRunConfig(\n    excluded_tags=[\"form\", \"header\", \"footer\"],\n    keep_data_attributes=False\n)\nresult = await crawler.arun(\"https://example.com\",…", "filename": ""}], "chunk_position": 40, "heading_path": "2. HTML Variants > 2. HTML Variants", "breadcrumbs": "Crawl Result and Output > 2. HTML Variants > 2. HTML Variants"}, {"id": "90c6ca1b58b61636", "url": "https://docs.crawl4ai.com/core/crawler-result/", "page_title": "Crawl Result and Output", "page_type": "api", "page_summary": "Describes the CrawlResult object returned by Crawl4AI's arun() method, including all fields, markdown generation, structured extraction, and additional outputs like links, media, tables, screenshots,…", "heading": "3. Markdown Generation", "content": "Page: Crawl Result and Output\nSection: 3. Markdown Generation\n\n###### 3.1 `markdown`\n\n- **`markdown`**: The current location for detailed markdown output, returning a **`MarkdownGenerationResult`** object.\n- **`markdown_v2`**: Removed in v0.5. Accessing it now…", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai.markdown_generation_strategy import DefaultMarkdownGenerator\n\nconfig = CrawlerRunConfig(…", "filename": ""}], "chunk_position": 40, "heading_path": "3. Markdown Generation > 3. Markdown Generation", "breadcrumbs": "Crawl Result and Output > 3. Markdown Generation > 3. Markdown Generation"}, {"id": "9d453eb37320a9ab", "url": "https://docs.crawl4ai.com/core/crawler-result/", "page_title": "Crawl Result and Output", "page_type": "api", "page_summary": "Describes the CrawlResult object returned by Crawl4AI's arun() method, including all fields, markdown generation, structured extraction, and additional outputs like links, media, tables, screenshots,…", "heading": "4. Structured Extraction: `extracted_content`", "content": "Page: Crawl Result and Output\nSection: 4. Structured Extraction: `extracted_content`\n\nIf you run a JSON-based extraction strategy (CSS, XPath, LLM, etc.), the structured data is **not** stored in `markdown`—it’s placed in **`result.extracted_content`** as a JSON string (or sometimes…", "code_blocks": [{"language": "python", "code": "import asyncio\nimport json\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\nfrom crawl4ai import JsonCssExtractionStrategy\n\nasync def main():\n    schema = {\n        \"name\": \"Example…", "filename": ""}], "chunk_position": 40, "heading_path": "4. Structured Extraction: `extracted_content` > 4. Structured Extraction: `extracted_content`", "breadcrumbs": "Crawl Result and Output > 4. Structured Extraction: `extracted_content` > 4. Structured Extraction: `extracted_content`"}, {"id": "95814b66ef3db0be", "url": "https://docs.crawl4ai.com/core/crawler-result/", "page_title": "Crawl Result and Output", "page_type": "api", "page_summary": "Describes the CrawlResult object returned by Crawl4AI's arun() method, including all fields, markdown generation, structured extraction, and additional outputs like links, media, tables, screenshots,…", "heading": "5. More Fields: Links, Media, Tables and More", "content": "Page: Crawl Result and Output\nSection: 5. More Fields: Links, Media, Tables and More\n\n###### 5.1 `links`\n\nA dictionary, typically with `\"internal\"` and `\"external\"` lists. Each entry might have `href`, `text`, `title`, etc. This is automatically captured if you haven’t disabled link…", "code_blocks": [{"language": "python", "code": "print(result.links[\"internal\"][:3])  # Show first 3 internal links", "filename": ""}, {"language": "python", "code": "images = result.media.get(\"images\", [])\nfor img in images:\n    print(\"Image URL:\", img[\"src\"], \"Alt:\", img.get(\"alt\"))", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nasync def main():\n    async with AsyncWebCrawler() as crawler:\n        result = await crawler.arun(…", "filename": ""}, {"language": "python", "code": "config = CrawlerRunConfig(\n    table_score_threshold=5  # Lower value = more tables detected (default: 7)\n)", "filename": ""}, {"language": "python", "code": "# Save the PDF\nwith open(\"page.pdf\", \"wb\") as f:\n    f.write(result.pdf)\n\n# Save the MHTML\nif result.mhtml:\n    with open(\"page.mhtml\", \"w\", encoding=\"utf-8\") as f:\n        f.write(result.mhtml)", "filename": ""}], "chunk_position": 40, "heading_path": "5. More Fields: Links, Media, Tables and More > 5. More Fields: Links, Media, Tables and More", "breadcrumbs": "Crawl Result and Output > 5. More Fields: Links, Media, Tables and More > 5. More Fields: Links, Media, Tables and More"}, {"id": "38737dda5be7532d", "url": "https://docs.crawl4ai.com/core/crawler-result/", "page_title": "Crawl Result and Output", "page_type": "api", "page_summary": "Describes the CrawlResult object returned by Crawl4AI's arun() method, including all fields, markdown generation, structured extraction, and additional outputs like links, media, tables, screenshots,…", "heading": "6. Accessing These Fields", "content": "Page: Crawl Result and Output\nSection: 6. Accessing These Fields\n\nAfter you run:\n\n```\nresult = await crawler.arun(url=\"https://example.com\", config=some_config)\n```\n\nCheck any field:\n\n```\nif result.success:\n    print(result.status_code, result.response_headers)…", "code_blocks": [{"language": "python", "code": "result = await crawler.arun(url=\"https://example.com\", config=some_config)", "filename": ""}, {"language": "python", "code": "if result.success:\n    print(result.status_code, result.response_headers)\n    print(\"Links found:\", len(result.links.get(\"internal\", [])))\n    if result.markdown:\n        print(\"Markdown snippet:\",…", "filename": ""}], "chunk_position": 40, "heading_path": "6. Accessing These Fields > 6. Accessing These Fields", "breadcrumbs": "Crawl Result and Output > 6. Accessing These Fields > 6. Accessing These Fields"}, {"id": "382470fbec65ccef", "url": "https://docs.crawl4ai.com/core/crawler-result/", "page_title": "Crawl Result and Output", "page_type": "api", "page_summary": "Describes the CrawlResult object returned by Crawl4AI's arun() method, including all fields, markdown generation, structured extraction, and additional outputs like links, media, tables, screenshots,…", "heading": "7. Next Steps", "content": "Page: Crawl Result and Output\nSection: 7. Next Steps\n\n- **Markdown Generation**: Dive deeper into how to configure `DefaultMarkdownGenerator` and various filters.\n- **Content Filtering**: Learn how to use `BM25ContentFilter` and…", "code_blocks": [], "chunk_position": 40, "heading_path": "7. Next Steps > 7. Next Steps", "breadcrumbs": "Crawl Result and Output > 7. Next Steps > 7. Next Steps"}, {"id": "77679d35b8f0fb18", "url": "https://docs.crawl4ai.com/core/deep-crawling/", "page_title": "Deep Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial explains how to perform configurable deep crawling with Crawl4AI, covering BFS, DFS, and BestFirst strategies, streaming vs non-streaming results, filters, scorers, crash recovery,…", "heading": "Introduction", "content": "Page: Deep Crawling - Crawl4AI Documentation (v0.9.x)\nSection: Introduction\n\nOne of Crawl4AI's most powerful features is its ability to perform **configurable deep crawling** that can explore websites beyond a single page. With fine-tuned control over crawl depth, domain…", "code_blocks": [], "chunk_position": 41, "heading_path": "Introduction > Introduction", "breadcrumbs": "Deep Crawling - Crawl4AI Documentation (v0.9.x) > Introduction > Introduction"}, {"id": "e1443db37c888c01", "url": "https://docs.crawl4ai.com/core/deep-crawling/", "page_title": "Deep Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial explains how to perform configurable deep crawling with Crawl4AI, covering BFS, DFS, and BestFirst strategies, streaming vs non-streaming results, filters, scorers, crash recovery,…", "heading": "1. Quick Example", "content": "Page: Deep Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 1. Quick Example\n\nHere's a minimal code snippet that implements a basic deep crawl using the **BFSDeepCrawlStrategy**:\n\n**What's happening?** \n- `BFSDeepCrawlStrategy(max_depth=2, include_external=False)` instructs…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai.deep_crawling import BFSDeepCrawlStrategy\nfrom crawl4ai.content_scraping_strategy import…", "filename": ""}], "chunk_position": 41, "heading_path": "1. Quick Example > 1. Quick Example", "breadcrumbs": "Deep Crawling - Crawl4AI Documentation (v0.9.x) > 1. Quick Example > 1. Quick Example"}, {"id": "33ecabea784bf741", "url": "https://docs.crawl4ai.com/core/deep-crawling/", "page_title": "Deep Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial explains how to perform configurable deep crawling with Crawl4AI, covering BFS, DFS, and BestFirst strategies, streaming vs non-streaming results, filters, scorers, crash recovery,…", "heading": "2. Understanding Deep Crawling Strategy Options", "content": "Page: Deep Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 2. Understanding Deep Crawling Strategy Options\n\n###### 2.1 BFSDeepCrawlStrategy (Breadth-First Search)\n\nThe **BFSDeepCrawlStrategy** uses a breadth-first approach, exploring all links at one depth before moving deeper:\n\n**Key parameters:** \n-…", "code_blocks": [{"language": "python", "code": "from crawl4ai.deep_crawling import BFSDeepCrawlStrategy\n\n# Basic configuration\nstrategy = BFSDeepCrawlStrategy(\n    max_depth=2,               # Crawl initial page + 2 levels deep…", "filename": ""}, {"language": "python", "code": "from crawl4ai.deep_crawling import DFSDeepCrawlStrategy\n\n# Basic configuration\nstrategy = DFSDeepCrawlStrategy(\n    max_depth=2,               # Crawl initial page + 2 levels deep…", "filename": ""}, {"language": "python", "code": "from crawl4ai.deep_crawling import BestFirstCrawlingStrategy\nfrom crawl4ai.deep_crawling.scorers import KeywordRelevanceScorer\n\n# Create a scorer\nscorer = KeywordRelevanceScorer(…", "filename": ""}], "chunk_position": 41, "heading_path": "2. Understanding Deep Crawling Strategy Options > 2. Understanding Deep Crawling Strategy Options", "breadcrumbs": "Deep Crawling - Crawl4AI Documentation (v0.9.x) > 2. Understanding Deep Crawling Strategy Options > 2. Understanding Deep Crawling Strategy Options"}, {"id": "137cd408d01e677f", "url": "https://docs.crawl4ai.com/core/deep-crawling/", "page_title": "Deep Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial explains how to perform configurable deep crawling with Crawl4AI, covering BFS, DFS, and BestFirst strategies, streaming vs non-streaming results, filters, scorers, crash recovery,…", "heading": "3. Streaming vs. Non-Streaming Results", "content": "Page: Deep Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 3. Streaming vs. Non-Streaming Results\n\nCrawl4AI can return results in two modes:\n\n###### 3.1 Non-Streaming Mode (Default)\n\n**When to use non-streaming mode:** \n- You need the complete dataset before processing\n- You're performing batch…", "code_blocks": [{"language": "python", "code": "config = CrawlerRunConfig(\n    deep_crawl_strategy=BFSDeepCrawlStrategy(max_depth=1),\n    stream=False  # Default behavior\n)\n\nasync with AsyncWebCrawler() as crawler:\n    # Wait for ALL results to be…", "filename": ""}, {"language": "python", "code": "config = CrawlerRunConfig(\n    deep_crawl_strategy=BFSDeepCrawlStrategy(max_depth=1),\n    stream=True  # Enable streaming\n)\n\nasync with AsyncWebCrawler() as crawler:\n    # Returns an async iterator…", "filename": ""}], "chunk_position": 41, "heading_path": "3. Streaming vs. Non-Streaming Results > 3. Streaming vs. Non-Streaming Results", "breadcrumbs": "Deep Crawling - Crawl4AI Documentation (v0.9.x) > 3. Streaming vs. Non-Streaming Results > 3. Streaming vs. Non-Streaming Results"}, {"id": "19ff0e6d6d062b4d", "url": "https://docs.crawl4ai.com/core/deep-crawling/", "page_title": "Deep Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial explains how to perform configurable deep crawling with Crawl4AI, covering BFS, DFS, and BestFirst strategies, streaming vs non-streaming results, filters, scorers, crash recovery,…", "heading": "4. Filtering Content with Filter Chains", "content": "Page: Deep Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 4. Filtering Content with Filter Chains\n\nFilters help you narrow down which pages to crawl. Combine multiple filters using **FilterChain** for powerful targeting.\n\n###### 4.1 Basic URL Pattern Filter\n\n###### 4.2 Combining Multiple Filters\n\n###…", "code_blocks": [{"language": "python", "code": "from crawl4ai.deep_crawling.filters import FilterChain, URLPatternFilter\n\n# Only follow URLs containing \"blog\" or \"docs\"\nurl_filter = URLPatternFilter(patterns=[\"*blog*\", \"*docs*\"])\n\nconfig =…", "filename": ""}, {"language": "python", "code": "from crawl4ai.deep_crawling.filters import (\n    FilterChain,\n    URLPatternFilter,\n    DomainFilter,\n    ContentTypeFilter\n)\n\n# Create a chain of filters\nfilter_chain = FilterChain([\n    # Only…", "filename": ""}], "chunk_position": 41, "heading_path": "4. Filtering Content with Filter Chains > 4. Filtering Content with Filter Chains", "breadcrumbs": "Deep Crawling - Crawl4AI Documentation (v0.9.x) > 4. Filtering Content with Filter Chains > 4. Filtering Content with Filter Chains"}, {"id": "10a7c4c890185855", "url": "https://docs.crawl4ai.com/core/deep-crawling/", "page_title": "Deep Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial explains how to perform configurable deep crawling with Crawl4AI, covering BFS, DFS, and BestFirst strategies, streaming vs non-streaming results, filters, scorers, crash recovery,…", "heading": "5. Using Scorers for Prioritized Crawling", "content": "Page: Deep Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 5. Using Scorers for Prioritized Crawling\n\nScorers assign priority values to discovered URLs, helping the crawler focus on the most relevant content first.\n\n###### 5.1 KeywordRelevanceScorer\n\n**How scorers work:** \n- Evaluate each discovered URL…", "code_blocks": [{"language": "python", "code": "from crawl4ai.deep_crawling.scorers import KeywordRelevanceScorer\nfrom crawl4ai.deep_crawling import BestFirstCrawlingStrategy\n\n# Create a keyword relevance scorer\nkeyword_scorer =…", "filename": ""}], "chunk_position": 41, "heading_path": "5. Using Scorers for Prioritized Crawling > 5. Using Scorers for Prioritized Crawling", "breadcrumbs": "Deep Crawling - Crawl4AI Documentation (v0.9.x) > 5. Using Scorers for Prioritized Crawling > 5. Using Scorers for Prioritized Crawling"}, {"id": "2958572076b9995b", "url": "https://docs.crawl4ai.com/core/deep-crawling/", "page_title": "Deep Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial explains how to perform configurable deep crawling with Crawl4AI, covering BFS, DFS, and BestFirst strategies, streaming vs non-streaming results, filters, scorers, crash recovery,…", "heading": "6. Advanced Filtering Techniques", "content": "Page: Deep Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 6. Advanced Filtering Techniques\n\n###### 6.1 SEO Filter for Quality Assessment\n\nThe **SEOFilter** helps you identify pages with strong SEO characteristics:\n\n###### 6.2 Content Relevance Filter\n\nThe **ContentRelevanceFilter** analyzes the…", "code_blocks": [{"language": "python", "code": "from crawl4ai.deep_crawling.filters import FilterChain, SEOFilter\n\n# Create an SEO filter that looks for specific keywords in page metadata\nseo_filter = SEOFilter(\n    threshold=0.5,  # Minimum score…", "filename": ""}, {"language": "python", "code": "from crawl4ai.deep_crawling.filters import FilterChain, ContentRelevanceFilter\n\n# Create a content relevance filter\nrelevance_filter = ContentRelevanceFilter(\n    query=\"Web crawling and data…", "filename": ""}], "chunk_position": 41, "heading_path": "6. Advanced Filtering Techniques > 6. Advanced Filtering Techniques", "breadcrumbs": "Deep Crawling - Crawl4AI Documentation (v0.9.x) > 6. Advanced Filtering Techniques > 6. Advanced Filtering Techniques"}, {"id": "9e311367c364b643", "url": "https://docs.crawl4ai.com/core/deep-crawling/", "page_title": "Deep Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial explains how to perform configurable deep crawling with Crawl4AI, covering BFS, DFS, and BestFirst strategies, streaming vs non-streaming results, filters, scorers, crash recovery,…", "heading": "7. Building a Complete Advanced Crawler", "content": "Page: Deep Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 7. Building a Complete Advanced Crawler\n\nThis example combines multiple techniques for a sophisticated crawl:", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai.content_scraping_strategy import LXMLWebScrapingStrategy\nfrom crawl4ai.deep_crawling import…", "filename": ""}], "chunk_position": 41, "heading_path": "7. Building a Complete Advanced Crawler > 7. Building a Complete Advanced Crawler", "breadcrumbs": "Deep Crawling - Crawl4AI Documentation (v0.9.x) > 7. Building a Complete Advanced Crawler > 7. Building a Complete Advanced Crawler"}, {"id": "a2905649842b3e82", "url": "https://docs.crawl4ai.com/core/deep-crawling/", "page_title": "Deep Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial explains how to perform configurable deep crawling with Crawl4AI, covering BFS, DFS, and BestFirst strategies, streaming vs non-streaming results, filters, scorers, crash recovery,…", "heading": "8. Limiting and Controlling Crawl Size", "content": "Page: Deep Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 8. Limiting and Controlling Crawl Size\n\n###### 8.1 Using max_pages\n\nYou can limit the total number of pages crawled with the `max_pages` parameter:\n\nThis feature is useful for:\n- Controlling API costs\n- Setting predictable execution times\n-…", "code_blocks": [{"language": "python", "code": "# Limit to exactly 20 pages regardless of depth\nstrategy = BFSDeepCrawlStrategy(\n    max_depth=3,\n    max_pages=20\n)", "filename": ""}, {"language": "python", "code": "# Only follow links with scores above 0.4\nstrategy = DFSDeepCrawlStrategy(\n    max_depth=2,\n    url_scorer=KeywordRelevanceScorer(keywords=[\"api\", \"guide\", \"reference\"]),\n    score_threshold=0.4  #…", "filename": ""}], "chunk_position": 41, "heading_path": "8. Limiting and Controlling Crawl Size > 8. Limiting and Controlling Crawl Size", "breadcrumbs": "Deep Crawling - Crawl4AI Documentation (v0.9.x) > 8. Limiting and Controlling Crawl Size > 8. Limiting and Controlling Crawl Size"}, {"id": "1f3bd945ce3455b0", "url": "https://docs.crawl4ai.com/core/deep-crawling/", "page_title": "Deep Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial explains how to perform configurable deep crawling with Crawl4AI, covering BFS, DFS, and BestFirst strategies, streaming vs non-streaming results, filters, scorers, crash recovery,…", "heading": "9. Common Pitfalls & Tips", "content": "Page: Deep Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 9. Common Pitfalls & Tips\n\n1. **Set realistic limits.** Be cautious with `max_depth` values > 3, which can exponentially increase crawl size. Use `max_pages` to set hard limits.\n\n2. **Don't neglect the scoring component.**…", "code_blocks": [{"language": "python", "code": "config = CrawlerRunConfig(\n    deep_crawl_strategy=BFSDeepCrawlStrategy(max_depth=2),\n    preserve_https_for_internal_links=True  # Keep HTTPS even if server redirects to HTTP\n)", "filename": ""}], "chunk_position": 41, "heading_path": "9. Common Pitfalls & Tips > 9. Common Pitfalls & Tips", "breadcrumbs": "Deep Crawling - Crawl4AI Documentation (v0.9.x) > 9. Common Pitfalls & Tips > 9. Common Pitfalls & Tips"}, {"id": "608331e47fba8e80", "url": "https://docs.crawl4ai.com/core/deep-crawling/", "page_title": "Deep Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial explains how to perform configurable deep crawling with Crawl4AI, covering BFS, DFS, and BestFirst strategies, streaming vs non-streaming results, filters, scorers, crash recovery,…", "heading": "10. Crash Recovery for Long-Running Crawls", "content": "Page: Deep Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 10. Crash Recovery for Long-Running Crawls\n\nFor production deployments, especially in cloud environments where instances can be terminated unexpectedly, Crawl4AI provides built-in crash recovery support for all deep crawl strategies.\n\n###### 10.1…", "code_blocks": [{"language": "python", "code": "from crawl4ai.deep_crawling import BFSDeepCrawlStrategy\nimport json\n\n# Callback to save state after each URL\nasync def save_state_to_redis(state: dict):\n    await redis.set(\"crawl_state\",…", "filename": ""}, {"language": "json", "code": "{\n    \"strategy_type\": \"bfs\",  # or \"dfs\", \"best_first\"\n    \"visited\": [\"url1\", \"url2\", ...],  # Already crawled URLs\n    \"pending\": [{\"url\": \"...\", \"parent_url\": \"...\"}],  # Queue/stack…", "filename": ""}, {"language": "python", "code": "import json\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai.deep_crawling import BFSDeepCrawlStrategy\n\n# Load saved state (e.g., from Redis, database, or file)\nsaved_state =…", "filename": ""}, {"language": "python", "code": "import json\n\ncaptured_state = None\n\nasync def capture_state(state: dict):\n    global captured_state\n    captured_state = state\n\nstrategy = BFSDeepCrawlStrategy(\n    max_depth=2,…", "filename": ""}, {"language": "python", "code": "import asyncio\nimport json\nimport redis.asyncio as redis\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai.deep_crawling import BFSDeepCrawlStrategy\n\nREDIS_KEY =…", "filename": ""}], "chunk_position": 41, "heading_path": "10. Crash Recovery for Long-Running Crawls > 10. Crash Recovery for Long-Running Crawls", "breadcrumbs": "Deep Crawling - Crawl4AI Documentation (v0.9.x) > 10. Crash Recovery for Long-Running Crawls > 10. Crash Recovery for Long-Running Crawls"}, {"id": "8053f530a0f5b597", "url": "https://docs.crawl4ai.com/core/deep-crawling/", "page_title": "Deep Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial explains how to perform configurable deep crawling with Crawl4AI, covering BFS, DFS, and BestFirst strategies, streaming vs non-streaming results, filters, scorers, crash recovery,…", "heading": "11. Cancellation Support for Deep Crawls", "content": "Page: Deep Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 11. Cancellation Support for Deep Crawls\n\nFor production environments like cloud platforms, you often need to stop a running crawl mid-execution—whether the user changed their mind, specified the wrong URL, or wants to control costs.…", "code_blocks": [{"language": "python", "code": "from crawl4ai.deep_crawling import BFSDeepCrawlStrategy\n\nasync def check_if_cancelled():\n    # Check Redis, database, or any external source\n    job = await redis.get(f\"job:{job_id}\")\n    return…", "filename": ""}, {"language": "python", "code": "strategy = BFSDeepCrawlStrategy(max_depth=3, max_pages=1000)\n\n# In another coroutine or thread:\nstrategy.cancel()  # Thread-safe, stops before next URL", "filename": ""}, {"language": "python", "code": "async with AsyncWebCrawler() as crawler:\n    results = await crawler.arun(url, config=config)\n\nif strategy.cancelled:\n    print(f\"Crawl was cancelled after {len(results)} pages\")\nelse:…", "filename": ""}, {"language": "python", "code": "async def handle_state(state: dict):\n    if state.get(\"cancelled\"):\n        print(\"Crawl was cancelled!\")\n        print(f\"Crawled {state['pages_crawled']} pages before cancellation\")\n    # Save state…", "filename": ""}, {"language": "python", "code": "import asyncio\nimport json\nimport redis.asyncio as redis\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai.deep_crawling import BFSDeepCrawlStrategy\n\nasync def…", "filename": ""}], "chunk_position": 41, "heading_path": "11. Cancellation Support for Deep Crawls > 11. Cancellation Support for Deep Crawls", "breadcrumbs": "Deep Crawling - Crawl4AI Documentation (v0.9.x) > 11. Cancellation Support for Deep Crawls > 11. Cancellation Support for Deep Crawls"}, {"id": "bd248215808c8155", "url": "https://docs.crawl4ai.com/core/deep-crawling/", "page_title": "Deep Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial explains how to perform configurable deep crawling with Crawl4AI, covering BFS, DFS, and BestFirst strategies, streaming vs non-streaming results, filters, scorers, crash recovery,…", "heading": "12. Prefetch Mode for Fast URL Discovery", "content": "Page: Deep Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 12. Prefetch Mode for Fast URL Discovery\n\nWhen you need to quickly discover URLs without full page processing, use **prefetch mode** . This is ideal for two-phase crawling where you first map the site, then selectively process specific…", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nconfig = CrawlerRunConfig(prefetch=True)\n\nasync with AsyncWebCrawler() as crawler:\n    result = await crawler.arun(\"https://example.com\",…", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nasync def two_phase_crawl(start_url: str):\n    async with AsyncWebCrawler() as crawler:\n        #…", "filename": ""}], "chunk_position": 41, "heading_path": "12. Prefetch Mode for Fast URL Discovery > 12. Prefetch Mode for Fast URL Discovery", "breadcrumbs": "Deep Crawling - Crawl4AI Documentation (v0.9.x) > 12. Prefetch Mode for Fast URL Discovery > 12. Prefetch Mode for Fast URL Discovery"}, {"id": "a4031a7c688379a2", "url": "https://docs.crawl4ai.com/core/deep-crawling/", "page_title": "Deep Crawling - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial explains how to perform configurable deep crawling with Crawl4AI, covering BFS, DFS, and BestFirst strategies, streaming vs non-streaming results, filters, scorers, crash recovery,…", "heading": "13. Summary & Next Steps", "content": "Page: Deep Crawling - Crawl4AI Documentation (v0.9.x)\nSection: 13. Summary & Next Steps\n\nIn this **Deep Crawling with Crawl4AI** tutorial, you learned to:\n\n- Configure **BFSDeepCrawlStrategy** , **DFSDeepCrawlStrategy** , and **BestFirstCrawlingStrategy**\n- Process results in streaming…", "code_blocks": [], "chunk_position": 41, "heading_path": "13. Summary & Next Steps > 13. Summary & Next Steps", "breadcrumbs": "Deep Crawling - Crawl4AI Documentation (v0.9.x) > 13. Summary & Next Steps > 13. Summary & Next Steps"}, {"id": "349ee17f5f52e86a", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "What Is Domain Mapping?", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: What Is Domain Mapping?\n\nDomain mapping goes beyond URL seeding. Instead of checking a single sitemap or index, `DomainMapper` combines **8 discovery sources** to find every URL under a domain — including subdomains you…", "code_blocks": [], "chunk_position": 42, "heading_path": "What Is Domain Mapping? > What Is Domain Mapping?", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > What Is Domain Mapping? > What Is Domain Mapping?"}, {"id": "3a09f69ad536f6f1", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "DomainMapper vs AsyncUrlSeeder", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: DomainMapper vs AsyncUrlSeeder\n\n| Aspect | AsyncUrlSeeder | DomainMapper |\n| --- | --- | --- |\n| **Scope** | Single host, listed URLs only | Entire domain + all subdomains |\n| **Sources** | Sitemap + Common Crawl | 8 sources…", "code_blocks": [], "chunk_position": 42, "heading_path": "DomainMapper vs AsyncUrlSeeder > DomainMapper vs AsyncUrlSeeder", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > DomainMapper vs AsyncUrlSeeder > DomainMapper vs AsyncUrlSeeder"}, {"id": "687e38fd283c409c", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "The 8 Discovery Sources", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: The 8 Discovery Sources\n\nDomainMapper combines these sources, each catching URLs the others miss:", "code_blocks": [], "chunk_position": 42, "heading_path": "The 8 Discovery Sources > The 8 Discovery Sources", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > The 8 Discovery Sources > The 8 Discovery Sources"}, {"id": "a2d8334c4513b01f", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "1. `sitemap` — Sitemap Discovery", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: 1. `sitemap` — Sitemap Discovery\n\nChecks `/sitemap.xml`, `/sitemap_index.xml`, and `robots.txt` `Sitemap:` directives **on every discovered host** — not just the root domain.", "code_blocks": [{"language": "python", "code": "config = DomainMapperConfig(source=\"sitemap\")", "filename": ""}], "chunk_position": 42, "heading_path": "1. `sitemap` — Sitemap Discovery > 1. `sitemap` — Sitemap Discovery", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > 1. `sitemap` — Sitemap Discovery > 1. `sitemap` — Sitemap Discovery"}, {"id": "16af11e917f63519", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "2. `cc` — Common Crawl", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: 2. `cc` — Common Crawl\n\nQueries the Common Crawl CDX API for `*.domain.tld/*`, catching URLs and subdomains the web's largest public crawl has indexed.", "code_blocks": [{"language": "python", "code": "config = DomainMapperConfig(source=\"cc\")", "filename": ""}], "chunk_position": 42, "heading_path": "2. `cc` — Common Crawl > 2. `cc` — Common Crawl", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > 2. `cc` — Common Crawl > 2. `cc` — Common Crawl"}, {"id": "877a95b21039565e", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "3. `wayback` — Wayback Machine", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: 3. `wayback` — Wayback Machine\n\nQueries the Internet Archive's CDX API. Often has different coverage than Common Crawl — including historical pages that have since been removed.", "code_blocks": [{"language": "python", "code": "config = DomainMapperConfig(source=\"wayback\")", "filename": ""}], "chunk_position": 42, "heading_path": "3. `wayback` — Wayback Machine > 3. `wayback` — Wayback Machine", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > 3. `wayback` — Wayback Machine > 3. `wayback` — Wayback Machine"}, {"id": "93fb7730a7d86dc4", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "4. `crt` — Certificate Transparency", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: 4. `crt` — Certificate Transparency\n\nQueries [crt.sh](https://crt.sh) for SSL certificates issued to `*.domain.tld`. This is the single most effective subdomain discovery technique — it found 14 subdomains for `superdesign.dev` that no…", "code_blocks": [{"language": "python", "code": "config = DomainMapperConfig(source=\"crt\")", "filename": ""}], "chunk_position": 42, "heading_path": "4. `crt` — Certificate Transparency > 4. `crt` — Certificate Transparency", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > 4. `crt` — Certificate Transparency > 4. `crt` — Certificate Transparency"}, {"id": "175ca22e33c319a6", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "5. `probe` — Common Path Probing", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: 5. `probe` — Common Path Probing\n\nTries ~25 well-known paths on each discovered host (`/docs`, `/api`, `/login`, `/dashboard`, `/openapi.json`, etc.). Combined with soft-404 detection to avoid false positives.", "code_blocks": [{"language": "python", "code": "config = DomainMapperConfig(source=\"probe\")\n\n# Add custom paths to probe\nconfig = DomainMapperConfig(\n    source=\"probe\",\n    probe_paths=[\"/custom-api\", \"/internal/status\"]\n)", "filename": ""}], "chunk_position": 42, "heading_path": "5. `probe` — Common Path Probing > 5. `probe` — Common Path Probing", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > 5. `probe` — Common Path Probing > 5. `probe` — Common Path Probing"}, {"id": "aae6cfcc1df1509e", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "6. `robots` — robots.txt Path Mining", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: 6. `robots` — robots.txt Path Mining\n\nParses `Disallow:` and `Allow:` lines from `robots.txt`. These are confirmed real paths the site acknowledges exist — often revealing admin panels, APIs, and internal tools that aren't linked…", "code_blocks": [{"language": "python", "code": "config = DomainMapperConfig(source=\"robots\")", "filename": ""}], "chunk_position": 42, "heading_path": "6. `robots` — robots.txt Path Mining > 6. `robots` — robots.txt Path Mining", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > 6. `robots` — robots.txt Path Mining > 6. `robots` — robots.txt Path Mining"}, {"id": "4053d70b77aa13e9", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "7. `feed` — RSS/Atom Feed Parsing", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: 7. `feed` — RSS/Atom Feed Parsing\n\nDiscovers and parses RSS/Atom feeds at common paths (`/feed`, `/rss`, `/atom.xml`, etc.). Feeds are curated lists of content URLs maintained by the site.", "code_blocks": [{"language": "python", "code": "config = DomainMapperConfig(source=\"feed\")", "filename": ""}], "chunk_position": 42, "heading_path": "7. `feed` — RSS/Atom Feed Parsing > 7. `feed` — RSS/Atom Feed Parsing", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > 7. `feed` — RSS/Atom Feed Parsing > 7. `feed` — RSS/Atom Feed Parsing"}, {"id": "64dcfe9173a5b925", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "8. `homepage` — Homepage Link Extraction", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: 8. `homepage` — Homepage Link Extraction\n\nFetches each host's homepage via HTTP and extracts all internal links using `quick_extract_links()`. Also mines `<link rel=\"alternate|preload|prefetch\">` tags from the `` for additional URLs.…", "code_blocks": [{"language": "python", "code": "config = DomainMapperConfig(source=\"homepage\")", "filename": ""}], "chunk_position": 42, "heading_path": "8. `homepage` — Homepage Link Extraction > 8. `homepage` — Homepage Link Extraction", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > 8. `homepage` — Homepage Link Extraction > 8. `homepage` — Homepage Link Extraction"}, {"id": "edcd3f97f49e3250", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "Phase 1: Host Discovery", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: Phase 1: Host Discovery\n\nDomainMapper first discovers all subdomains under your domain:\n\nEach discovered host is validated with an HTTP HEAD request. Hosts that don't respond are dropped.", "code_blocks": [{"language": "text", "code": "superdesign.dev\n├── crt.sh           → docs, app, cloud, insights, staging-api, ui2web, ...\n├── Wayback CDX      → api, app, docs, www, ...\n├── Common Crawl     → app, www, ...\n└── DNS guessing     →…", "filename": ""}], "chunk_position": 42, "heading_path": "Phase 1: Host Discovery > Phase 1: Host Discovery", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > Phase 1: Host Discovery > Phase 1: Host Discovery"}, {"id": "dd106ed0d70ec983", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "Phase 2: Per-Host Scanning", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: Phase 2: Per-Host Scanning\n\nFor each validated host, DomainMapper runs all enabled sources in parallel:", "code_blocks": [{"language": "text", "code": "docs.superdesign.dev\n├── Soft-404 fingerprint  → (404 returns proper error — no SPA issue)\n├── robots.txt            → 1 sitemap URL, 1 disallow path\n├── Sitemap parsing       → 19 URLs\n├── Path…", "filename": ""}], "chunk_position": 42, "heading_path": "Phase 2: Per-Host Scanning > Phase 2: Per-Host Scanning", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > Phase 2: Per-Host Scanning > Phase 2: Per-Host Scanning"}, {"id": "c7b6c7f4bbf64e46", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "Phase 3: Post-Processing", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: Phase 3: Post-Processing\n\nAll discovered URLs go through:\n\n- **URL normalization** — using `normalize_url()` to canonicalize\n- **Deduplication** — by normalized URL, merging source attribution\n- **Nonsense filtering** —…", "code_blocks": [], "chunk_position": 42, "heading_path": "Phase 3: Post-Processing > Phase 3: Post-Processing", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > Phase 3: Post-Processing > Phase 3: Post-Processing"}, {"id": "29cdd1388f1d1836", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "Soft-404 Detection", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: Soft-404 Detection\n\nMany modern SPAs return HTTP 200 for every URL — even pages that don't exist. DomainMapper detects this:\n\n- **Fingerprinting**: Fetches a guaranteed-nonexistent URL (e.g., `/c4ai-probe-a1b2c3d4`) on…", "code_blocks": [{"language": "python", "code": "# Soft-404 detection is on by default\nconfig = DomainMapperConfig(soft_404_detection=True)\n\n# Disable if you want raw results\nconfig = DomainMapperConfig(soft_404_detection=False)", "filename": ""}], "chunk_position": 42, "heading_path": "Soft-404 Detection > Soft-404 Detection", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > Soft-404 Detection > Soft-404 Detection"}, {"id": "e8789c66db4e7159", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "DomainMapperConfig", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: DomainMapperConfig\n\n| Parameter | Type | Default | Description |\n| --- | --- | --- | --- |\n| `source` | str | `\"sitemap+cc+crt+probe\"` | Discovery sources joined by `+` |\n| `max_urls` | int | `-1` | Maximum URLs to…", "code_blocks": [], "chunk_position": 42, "heading_path": "DomainMapperConfig > DomainMapperConfig", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > DomainMapperConfig > DomainMapperConfig"}, {"id": "57ee3e64b78d6fd6", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "Discover and Crawl Documentation", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: Discover and Crawl Documentation\n\n```python\nimport asyncio\nfrom crawl4ai import AsyncWebCrawler, DomainMapperConfig, CrawlerRunConfig\n\nasync def crawl_all_docs():\n    async with AsyncWebCrawler() as crawler:\n        # Step 1: Discover all…\n```", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, DomainMapperConfig, CrawlerRunConfig\n\nasync def crawl_all_docs():\n    async with AsyncWebCrawler() as crawler:\n        # Step 1: Discover all…", "filename": ""}], "chunk_position": 42, "heading_path": "Discover and Crawl Documentation > Discover and Crawl Documentation", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > Discover and Crawl Documentation > Discover and Crawl Documentation"}, {"id": "b519e1bedfe0dd6e", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "Security Audit: Find Exposed Services", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: Security Audit: Find Exposed Services\n\n```python\nasync def audit_domain():\n    async with DomainMapper() as mapper:\n        results = await mapper.scan(\"company.com\", DomainMapperConfig(\n            source=\"crt+probe+robots\",…\n```", "code_blocks": [{"language": "python", "code": "async def audit_domain():\n    async with DomainMapper() as mapper:\n        results = await mapper.scan(\"company.com\", DomainMapperConfig(\n            source=\"crt+probe+robots\",…", "filename": ""}], "chunk_position": 42, "heading_path": "Security Audit: Find Exposed Services > Security Audit: Find Exposed Services", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > Security Audit: Find Exposed Services > Security Audit: Find Exposed Services"}, {"id": "4e3779f4c6ae0c03", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "Compare Subdomains Across a Domain", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: Compare Subdomains Across a Domain\n\n```python\nasync def map_infrastructure():\n    async with DomainMapper() as mapper:\n        results = await mapper.scan(\"company.com\", DomainMapperConfig(\n            source=\"crt+probe\",…\n```", "code_blocks": [{"language": "python", "code": "async def map_infrastructure():\n    async with DomainMapper() as mapper:\n        results = await mapper.scan(\"company.com\", DomainMapperConfig(\n            source=\"crt+probe\",…", "filename": ""}], "chunk_position": 42, "heading_path": "Compare Subdomains Across a Domain > Compare Subdomains Across a Domain", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > Compare Subdomains Across a Domain > Compare Subdomains Across a Domain"}, {"id": "39312523d0b4caf8", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "Tips and Best Practices", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: Tips and Best Practices\n\n- **Start with the default sources** (`sitemap+cc+crt+probe`). Add `wayback`, `robots`, `feed`, and `homepage` if you need maximum coverage.\n- **Use `extract_head=False` for speed** when you just…", "code_blocks": [], "chunk_position": 42, "heading_path": "Tips and Best Practices > Tips and Best Practices", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > Tips and Best Practices > Tips and Best Practices"}, {"id": "9ff671a6a9c24233", "url": "https://docs.crawl4ai.com/core/domain-mapping/", "page_title": "Domain Mapping: Discover Every URL Under a Domain", "page_type": "reference", "page_summary": "This page describes DomainMapper in Crawl4AI, which discovers every URL under a domain using 8 discovery sources, including sitemaps, Common Crawl, Wayback Machine, Certificate Transparency, path…", "heading": "See Also", "content": "Page: Domain Mapping: Discover Every URL Under a Domain\nSection: See Also\n\n- [URL Seeding](../url-seeding/) — simpler, single-host URL discovery from sitemaps and Common Crawl\n- [Deep Crawling](../deep-crawling/) — follow links dynamically within pages\n- [Multi-URL…", "code_blocks": [], "chunk_position": 42, "heading_path": "See Also > See Also", "breadcrumbs": "Domain Mapping: Discover Every URL Under a Domain > See Also > See Also"}, {"id": "71311c098e614957", "url": "https://docs.crawl4ai.com/core/examples/", "page_title": "Code Examples", "page_type": "overview", "page_summary": "This page provides a comprehensive list of example scripts that demonstrate various features and capabilities of Crawl4AI, organized by category with links to code and guides.", "heading": "Overview", "content": "Page: Code Examples\nSection: Overview\n\nThis page provides a comprehensive list of example scripts that demonstrate various features and capabilities of Crawl4AI. Each example is designed to showcase specific functionality, making it…", "code_blocks": [], "chunk_position": 43, "heading_path": "Overview > Overview", "breadcrumbs": "Code Examples > Overview > Overview"}, {"id": "dcf22870800953b9", "url": "https://docs.crawl4ai.com/core/examples/", "page_title": "Code Examples", "page_type": "overview", "page_summary": "This page provides a comprehensive list of example scripts that demonstrate various features and capabilities of Crawl4AI, organized by category with links to code and guides.", "heading": "Getting Started Examples", "content": "Page: Code Examples\nSection: Getting Started Examples\n\n| Example | Description | Link |\n| --- | --- | --- |\n| Hello World | A simple introductory example demonstrating basic usage of AsyncWebCrawler with JavaScript execution and content filtering. |…", "code_blocks": [], "chunk_position": 43, "heading_path": "Getting Started Examples > Getting Started Examples", "breadcrumbs": "Code Examples > Getting Started Examples > Getting Started Examples"}, {"id": "6212c11c80ddcbfc", "url": "https://docs.crawl4ai.com/core/examples/", "page_title": "Code Examples", "page_type": "overview", "page_summary": "This page provides a comprehensive list of example scripts that demonstrate various features and capabilities of Crawl4AI, organized by category with links to code and guides.", "heading": "Proxies", "content": "Page: Code Examples\nSection: Proxies\n\n| Example | Description | Link |\n| --- | --- | --- |\n| **NSTProxy** | [NSTProxy](https://www.nstproxy.com/?utm_source=crawl4ai) Seamlessly integrates with crawl4ai — no setup required. Access…", "code_blocks": [], "chunk_position": 43, "heading_path": "Proxies > Proxies", "breadcrumbs": "Code Examples > Proxies > Proxies"}, {"id": "db21399ac61b2f8f", "url": "https://docs.crawl4ai.com/core/examples/", "page_title": "Code Examples", "page_type": "overview", "page_summary": "This page provides a comprehensive list of example scripts that demonstrate various features and capabilities of Crawl4AI, organized by category with links to code and guides.", "heading": "Browser & Crawling Features", "content": "Page: Code Examples\nSection: Browser & Crawling Features\n\n| Example | Description | Link |\n| --- | --- | --- |\n| Built-in Browser | Demonstrates how to use the built-in browser capabilities. | [View…", "code_blocks": [], "chunk_position": 43, "heading_path": "Browser & Crawling Features > Browser & Crawling Features", "breadcrumbs": "Code Examples > Browser & Crawling Features > Browser & Crawling Features"}, {"id": "f0987ef8c214b9a0", "url": "https://docs.crawl4ai.com/core/examples/", "page_title": "Code Examples", "page_type": "overview", "page_summary": "This page provides a comprehensive list of example scripts that demonstrate various features and capabilities of Crawl4AI, organized by category with links to code and guides.", "heading": "Advanced Crawling & Deep Crawling", "content": "Page: Code Examples\nSection: Advanced Crawling & Deep Crawling\n\n| Example | Description | Link |\n| --- | --- | --- |\n| Deep Crawling | An extensive tutorial on deep crawling capabilities, demonstrating BFS and BestFirst strategies, stream vs. non-stream…", "code_blocks": [], "chunk_position": 43, "heading_path": "Advanced Crawling & Deep Crawling > Advanced Crawling & Deep Crawling", "breadcrumbs": "Code Examples > Advanced Crawling & Deep Crawling > Advanced Crawling & Deep Crawling"}, {"id": "908eca97ff7dbb77", "url": "https://docs.crawl4ai.com/core/examples/", "page_title": "Code Examples", "page_type": "overview", "page_summary": "This page provides a comprehensive list of example scripts that demonstrate various features and capabilities of Crawl4AI, organized by category with links to code and guides.", "heading": "Extraction Strategies", "content": "Page: Code Examples\nSection: Extraction Strategies\n\n| Example | Description | Link |\n| --- | --- | --- |\n| Extraction Strategies | Demonstrates different extraction strategies with various input formats (markdown, HTML, fit_markdown) and JSON-based…", "code_blocks": [], "chunk_position": 43, "heading_path": "Extraction Strategies > Extraction Strategies", "breadcrumbs": "Code Examples > Extraction Strategies > Extraction Strategies"}, {"id": "ca0193273e80aeec", "url": "https://docs.crawl4ai.com/core/examples/", "page_title": "Code Examples", "page_type": "overview", "page_summary": "This page provides a comprehensive list of example scripts that demonstrate various features and capabilities of Crawl4AI, organized by category with links to code and guides.", "heading": "E-commerce & Specialized Crawling", "content": "Page: Code Examples\nSection: E-commerce & Specialized Crawling\n\n| Example | Description | Link |\n| --- | --- | --- |\n| Amazon Product Extraction | Demonstrates how to extract structured product data from Amazon search results using CSS selectors. | [View…", "code_blocks": [], "chunk_position": 43, "heading_path": "E-commerce & Specialized Crawling > E-commerce & Specialized Crawling", "breadcrumbs": "Code Examples > E-commerce & Specialized Crawling > E-commerce & Specialized Crawling"}, {"id": "02fcc1f9f3165c2e", "url": "https://docs.crawl4ai.com/core/examples/", "page_title": "Code Examples", "page_type": "overview", "page_summary": "This page provides a comprehensive list of example scripts that demonstrate various features and capabilities of Crawl4AI, organized by category with links to code and guides.", "heading": "Anti-Bot & Stealth Features", "content": "Page: Code Examples\nSection: Anti-Bot & Stealth Features\n\n| Example | Description | Link |\n| --- | --- | --- |\n| Stealth Mode Quick Start | Five practical examples showing how to use stealth mode for bypassing basic bot detection. | [View…", "code_blocks": [], "chunk_position": 43, "heading_path": "Anti-Bot & Stealth Features > Anti-Bot & Stealth Features", "breadcrumbs": "Code Examples > Anti-Bot & Stealth Features > Anti-Bot & Stealth Features"}, {"id": "8edee6d27b7d2a43", "url": "https://docs.crawl4ai.com/core/examples/", "page_title": "Code Examples", "page_type": "overview", "page_summary": "This page provides a comprehensive list of example scripts that demonstrate various features and capabilities of Crawl4AI, organized by category with links to code and guides.", "heading": "Customization & Security", "content": "Page: Code Examples\nSection: Customization & Security\n\n| Example | Description | Link |\n| --- | --- | --- |\n| Hooks | Illustrates how to use hooks at different stages of the crawling process for advanced customization. | [View…", "code_blocks": [], "chunk_position": 43, "heading_path": "Customization & Security > Customization & Security", "breadcrumbs": "Code Examples > Customization & Security > Customization & Security"}, {"id": "6f24ecfd78055d85", "url": "https://docs.crawl4ai.com/core/examples/", "page_title": "Code Examples", "page_type": "overview", "page_summary": "This page provides a comprehensive list of example scripts that demonstrate various features and capabilities of Crawl4AI, organized by category with links to code and guides.", "heading": "Docker & Deployment", "content": "Page: Code Examples\nSection: Docker & Deployment\n\n| Example | Description | Link |\n| --- | --- | --- |\n| Docker Config | Demonstrates how to create and use Docker configuration objects. | [View…", "code_blocks": [], "chunk_position": 43, "heading_path": "Docker & Deployment > Docker & Deployment", "breadcrumbs": "Code Examples > Docker & Deployment > Docker & Deployment"}, {"id": "388a488d27748b9a", "url": "https://docs.crawl4ai.com/core/examples/", "page_title": "Code Examples", "page_type": "overview", "page_summary": "This page provides a comprehensive list of example scripts that demonstrate various features and capabilities of Crawl4AI, organized by category with links to code and guides.", "heading": "Application Examples", "content": "Page: Code Examples\nSection: Application Examples\n\n| Example | Description | Link |\n| --- | --- | --- |\n| Research Assistant | Demonstrates how to build a research assistant using Crawl4AI. | [View…", "code_blocks": [], "chunk_position": 43, "heading_path": "Application Examples > Application Examples", "breadcrumbs": "Code Examples > Application Examples > Application Examples"}, {"id": "d1a6f90961506aed", "url": "https://docs.crawl4ai.com/core/examples/", "page_title": "Code Examples", "page_type": "overview", "page_summary": "This page provides a comprehensive list of example scripts that demonstrate various features and capabilities of Crawl4AI, organized by category with links to code and guides.", "heading": "Content Generation & Markdown", "content": "Page: Code Examples\nSection: Content Generation & Markdown\n\n| Example | Description | Link |\n| --- | --- | --- |\n| Content Source | Demonstrates how to work with different content sources in markdown generation. | [View…", "code_blocks": [], "chunk_position": 43, "heading_path": "Content Generation & Markdown > Content Generation & Markdown", "breadcrumbs": "Code Examples > Content Generation & Markdown > Content Generation & Markdown"}, {"id": "4b26ee72eddf59b2", "url": "https://docs.crawl4ai.com/core/examples/", "page_title": "Code Examples", "page_type": "overview", "page_summary": "This page provides a comprehensive list of example scripts that demonstrate various features and capabilities of Crawl4AI, organized by category with links to code and guides.", "heading": "Running the Examples", "content": "Page: Code Examples\nSection: Running the Examples\n\nTo run any of these examples, you'll need to have Crawl4AI installed:\n\nThen, you can run an example script like this:\n\nFor examples that require additional dependencies or environment variables,…", "code_blocks": [{"language": "bash", "code": "pip install crawl4ai", "filename": ""}, {"language": "bash", "code": "python -m docs.examples.hello_world", "filename": ""}], "chunk_position": 43, "heading_path": "Running the Examples > Running the Examples", "breadcrumbs": "Code Examples > Running the Examples > Running the Examples"}, {"id": "bcfedbf734edea6c", "url": "https://docs.crawl4ai.com/core/examples/", "page_title": "Code Examples", "page_type": "overview", "page_summary": "This page provides a comprehensive list of example scripts that demonstrate various features and capabilities of Crawl4AI, organized by category with links to code and guides.", "heading": "Contributing New Examples", "content": "Page: Code Examples\nSection: Contributing New Examples\n\nIf you've created an interesting example that demonstrates a unique use case or feature of Crawl4AI, we encourage you to contribute it to our examples collection. Please see our [contribution…", "code_blocks": [], "chunk_position": 43, "heading_path": "Contributing New Examples > Contributing New Examples", "breadcrumbs": "Code Examples > Contributing New Examples > Contributing New Examples"}, {"id": "d0c0925157bb017f", "url": "https://docs.crawl4ai.com/core/fit-markdown/", "page_title": "Fit Markdown with Pruning & BM25", "page_type": "guide", "page_summary": "Explains how to use Fit Markdown with Pruning and BM25 content filters in Crawl4AI to extract concise, relevant content from web pages.", "heading": "Overview", "content": "Page: Fit Markdown with Pruning & BM25\nSection: Overview\n\n**Fit Markdown** is a specialized **filtered** version of your page’s markdown, focusing on the most relevant content. By default, Crawl4AI converts the entire HTML into a broad **raw_markdown**.…", "code_blocks": [], "chunk_position": 44, "heading_path": "Overview > Overview", "breadcrumbs": "Fit Markdown with Pruning & BM25 > Overview > Overview"}, {"id": "099e399fbb28610f", "url": "https://docs.crawl4ai.com/core/fit-markdown/", "page_title": "Fit Markdown with Pruning & BM25", "page_type": "guide", "page_summary": "Explains how to use Fit Markdown with Pruning and BM25 content filters in Crawl4AI to extract concise, relevant content from web pages.", "heading": "1.1 The `content_filter`", "content": "Page: Fit Markdown with Pruning & BM25\nSection: 1.1 The `content_filter`\n\nIn **`CrawlerRunConfig`**, you can specify a **`content_filter`** to shape how content is pruned or ranked before final markdown generation. A filter’s logic is applied **before** or **during** the…", "code_blocks": [], "chunk_position": 44, "heading_path": "1.1 The `content_filter` > 1.1 The `content_filter`", "breadcrumbs": "Fit Markdown with Pruning & BM25 > 1.1 The `content_filter` > 1.1 The `content_filter`"}, {"id": "98b9ec433bafc56d", "url": "https://docs.crawl4ai.com/core/fit-markdown/", "page_title": "Fit Markdown with Pruning & BM25", "page_type": "guide", "page_summary": "Explains how to use Fit Markdown with Pruning and BM25 content filters in Crawl4AI to extract concise, relevant content from web pages.", "heading": "1.2 Common Filters", "content": "Page: Fit Markdown with Pruning & BM25\nSection: 1.2 Common Filters\n\n1. **PruningContentFilter** – Scores each node by text density, link density, and tag importance, discarding those below a threshold.\n2. **BM25ContentFilter** – Focuses on textual relevance using…", "code_blocks": [], "chunk_position": 44, "heading_path": "1.2 Common Filters > 1.2 Common Filters", "breadcrumbs": "Fit Markdown with Pruning & BM25 > 1.2 Common Filters > 1.2 Common Filters"}, {"id": "40833dd1a0077d6d", "url": "https://docs.crawl4ai.com/core/fit-markdown/", "page_title": "Fit Markdown with Pruning & BM25", "page_type": "guide", "page_summary": "Explains how to use Fit Markdown with Pruning and BM25 content filters in Crawl4AI to extract concise, relevant content from web pages.", "heading": "2. PruningContentFilter", "content": "Page: Fit Markdown with Pruning & BM25\nSection: 2. PruningContentFilter\n\n**Pruning** discards less relevant nodes based on **text density, link density, and tag importance**. It’s a heuristic-based approach—if certain sections appear too “thin” or too “spammy,” they’re…", "code_blocks": [], "chunk_position": 44, "heading_path": "2. PruningContentFilter > 2. PruningContentFilter", "breadcrumbs": "Fit Markdown with Pruning & BM25 > 2. PruningContentFilter > 2. PruningContentFilter"}, {"id": "d12012630a60c4af", "url": "https://docs.crawl4ai.com/core/fit-markdown/", "page_title": "Fit Markdown with Pruning & BM25", "page_type": "guide", "page_summary": "Explains how to use Fit Markdown with Pruning and BM25 content filters in Crawl4AI to extract concise, relevant content from web pages.", "heading": "2.1 Usage Example", "content": "Page: Fit Markdown with Pruning & BM25\nSection: 2.1 Usage Example\n\n```python\nimport asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai.content_filter_strategy import PruningContentFilter\nfrom crawl4ai.markdown_generation_strategy import…\n```", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai.content_filter_strategy import PruningContentFilter\nfrom crawl4ai.markdown_generation_strategy import…", "filename": ""}], "chunk_position": 44, "heading_path": "2.1 Usage Example > 2.1 Usage Example", "breadcrumbs": "Fit Markdown with Pruning & BM25 > 2.1 Usage Example > 2.1 Usage Example"}, {"id": "6f4d9744491ce50b", "url": "https://docs.crawl4ai.com/core/fit-markdown/", "page_title": "Fit Markdown with Pruning & BM25", "page_type": "guide", "page_summary": "Explains how to use Fit Markdown with Pruning and BM25 content filters in Crawl4AI to extract concise, relevant content from web pages.", "heading": "2.2 Key Parameters", "content": "Page: Fit Markdown with Pruning & BM25\nSection: 2.2 Key Parameters\n\n- **`min_word_threshold`** (int): If a block has fewer words than this, it’s pruned.\n- **`threshold_type`** (str):\n  - `\"fixed\"` → each node must exceed `threshold` (0–1).\n  - `\"dynamic\"` → node…", "code_blocks": [], "chunk_position": 44, "heading_path": "2.2 Key Parameters > 2.2 Key Parameters", "breadcrumbs": "Fit Markdown with Pruning & BM25 > 2.2 Key Parameters > 2.2 Key Parameters"}, {"id": "34a4b97ccd2a31b4", "url": "https://docs.crawl4ai.com/core/fit-markdown/", "page_title": "Fit Markdown with Pruning & BM25", "page_type": "guide", "page_summary": "Explains how to use Fit Markdown with Pruning and BM25 content filters in Crawl4AI to extract concise, relevant content from web pages.", "heading": "3. BM25ContentFilter", "content": "Page: Fit Markdown with Pruning & BM25\nSection: 3. BM25ContentFilter\n\n**BM25** is a classical text ranking algorithm often used in search engines. If you have a **user query** or rely on page metadata to derive a query, BM25 can identify which text chunks best match…", "code_blocks": [], "chunk_position": 44, "heading_path": "3. BM25ContentFilter > 3. BM25ContentFilter", "breadcrumbs": "Fit Markdown with Pruning & BM25 > 3. BM25ContentFilter > 3. BM25ContentFilter"}, {"id": "ea9fcbb56557d379", "url": "https://docs.crawl4ai.com/core/fit-markdown/", "page_title": "Fit Markdown with Pruning & BM25", "page_type": "guide", "page_summary": "Explains how to use Fit Markdown with Pruning and BM25 content filters in Crawl4AI to extract concise, relevant content from web pages.", "heading": "3.1 Usage Example", "content": "Page: Fit Markdown with Pruning & BM25\nSection: 3.1 Usage Example\n\n```python\nimport asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai.content_filter_strategy import BM25ContentFilter\nfrom crawl4ai.markdown_generation_strategy import…\n```", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai.content_filter_strategy import BM25ContentFilter\nfrom crawl4ai.markdown_generation_strategy import…", "filename": ""}], "chunk_position": 44, "heading_path": "3.1 Usage Example > 3.1 Usage Example", "breadcrumbs": "Fit Markdown with Pruning & BM25 > 3.1 Usage Example > 3.1 Usage Example"}, {"id": "a5c024dcec085305", "url": "https://docs.crawl4ai.com/core/fit-markdown/", "page_title": "Fit Markdown with Pruning & BM25", "page_type": "guide", "page_summary": "Explains how to use Fit Markdown with Pruning and BM25 content filters in Crawl4AI to extract concise, relevant content from web pages.", "heading": "3.2 Parameters", "content": "Page: Fit Markdown with Pruning & BM25\nSection: 3.2 Parameters\n\n- **`user_query`** (str, optional): E.g. `\"machine learning\"`. If blank, the filter tries to glean a query from page metadata.\n- **`bm25_threshold`** (float, default 1.0):\n  - Higher → fewer chunks…", "code_blocks": [], "chunk_position": 44, "heading_path": "3.2 Parameters > 3.2 Parameters", "breadcrumbs": "Fit Markdown with Pruning & BM25 > 3.2 Parameters > 3.2 Parameters"}, {"id": "e113e6d57c485d76", "url": "https://docs.crawl4ai.com/core/fit-markdown/", "page_title": "Fit Markdown with Pruning & BM25", "page_type": "guide", "page_summary": "Explains how to use Fit Markdown with Pruning and BM25 content filters in Crawl4AI to extract concise, relevant content from web pages.", "heading": "4. Accessing the “Fit” Output", "content": "Page: Fit Markdown with Pruning & BM25\nSection: 4. Accessing the “Fit” Output\n\nAfter the crawl, your “fit” content is found in **`result.markdown.fit_markdown`**.\n\nIf the content filter is **BM25**, you might see additional logic or references in `fit_markdown` that highlight…", "code_blocks": [{"language": "python", "code": "fit_md = result.markdown.fit_markdown\nfit_html = result.markdown.fit_html", "filename": ""}], "chunk_position": 44, "heading_path": "4. Accessing the “Fit” Output > 4. Accessing the “Fit” Output", "breadcrumbs": "Fit Markdown with Pruning & BM25 > 4. Accessing the “Fit” Output > 4. Accessing the “Fit” Output"}, {"id": "63b8c9f0b2d0cf2c", "url": "https://docs.crawl4ai.com/core/fit-markdown/", "page_title": "Fit Markdown with Pruning & BM25", "page_type": "guide", "page_summary": "Explains how to use Fit Markdown with Pruning and BM25 content filters in Crawl4AI to extract concise, relevant content from web pages.", "heading": "5.1 Pruning", "content": "Page: Fit Markdown with Pruning & BM25\nSection: 5.1 Pruning\n\n```python\nprune_filter = PruningContentFilter(\n    threshold=0.5,\n    threshold_type=\"fixed\",\n    min_word_threshold=10\n)\nmd_generator = DefaultMarkdownGenerator(content_filter=prune_filter)\nconfig =…\n```", "code_blocks": [{"language": "python", "code": "prune_filter = PruningContentFilter(\n    threshold=0.5,\n    threshold_type=\"fixed\",\n    min_word_threshold=10\n)\nmd_generator = DefaultMarkdownGenerator(content_filter=prune_filter)\nconfig =…", "filename": ""}], "chunk_position": 44, "heading_path": "5.1 Pruning > 5.1 Pruning", "breadcrumbs": "Fit Markdown with Pruning & BM25 > 5.1 Pruning > 5.1 Pruning"}, {"id": "1262245e2424413d", "url": "https://docs.crawl4ai.com/core/fit-markdown/", "page_title": "Fit Markdown with Pruning & BM25", "page_type": "guide", "page_summary": "Explains how to use Fit Markdown with Pruning and BM25 content filters in Crawl4AI to extract concise, relevant content from web pages.", "heading": "5.2 BM25", "content": "Page: Fit Markdown with Pruning & BM25\nSection: 5.2 BM25\n\n```python\nbm25_filter = BM25ContentFilter(\n    user_query=\"health benefits fruit\",\n    bm25_threshold=1.2\n)\nmd_generator = DefaultMarkdownGenerator(content_filter=bm25_filter)\nconfig =…\n```", "code_blocks": [{"language": "python", "code": "bm25_filter = BM25ContentFilter(\n    user_query=\"health benefits fruit\",\n    bm25_threshold=1.2\n)\nmd_generator = DefaultMarkdownGenerator(content_filter=bm25_filter)\nconfig =…", "filename": ""}], "chunk_position": 44, "heading_path": "5.2 BM25 > 5.2 BM25", "breadcrumbs": "Fit Markdown with Pruning & BM25 > 5.2 BM25 > 5.2 BM25"}, {"id": "aeb743d9903cce55", "url": "https://docs.crawl4ai.com/core/fit-markdown/", "page_title": "Fit Markdown with Pruning & BM25", "page_type": "guide", "page_summary": "Explains how to use Fit Markdown with Pruning and BM25 content filters in Crawl4AI to extract concise, relevant content from web pages.", "heading": "6. Combining with “word_count_threshold” & Exclusions", "content": "Page: Fit Markdown with Pruning & BM25\nSection: 6. Combining with “word_count_threshold” & Exclusions\n\nRemember you can also specify:\n\nThus, **multi-level** filtering occurs:\n\n- The crawler’s `excluded_tags` are removed from the HTML first.\n- The content filter (Pruning, BM25, or custom) prunes or…", "code_blocks": [{"language": "python", "code": "config = CrawlerRunConfig(\n    word_count_threshold=10,\n    excluded_tags=[\"nav\", \"footer\", \"header\"],\n    exclude_external_links=True,\n    markdown_generator=DefaultMarkdownGenerator(…", "filename": ""}], "chunk_position": 44, "heading_path": "6. Combining with “word_count_threshold” & Exclusions > 6. Combining with “word_count_threshold” & Exclusions", "breadcrumbs": "Fit Markdown with Pruning & BM25 > 6. Combining with “word_count_threshold” & Exclusions > 6. Combining with “word_count_threshold” & Exclusions"}, {"id": "35cd6e6fbc3a8551", "url": "https://docs.crawl4ai.com/core/fit-markdown/", "page_title": "Fit Markdown with Pruning & BM25", "page_type": "guide", "page_summary": "Explains how to use Fit Markdown with Pruning and BM25 content filters in Crawl4AI to extract concise, relevant content from web pages.", "heading": "7. Custom Filters", "content": "Page: Fit Markdown with Pruning & BM25\nSection: 7. Custom Filters\n\nIf you need a different approach (like a specialized ML model or site-specific heuristics), you can create a new class inheriting from `RelevantContentFilter` and implement `filter_content(html)`.…", "code_blocks": [{"language": "python", "code": "from crawl4ai.content_filter_strategy import RelevantContentFilter\n\nclass MyCustomFilter(RelevantContentFilter):\n    def filter_content(self, html, min_word_threshold=None):\n        # parse HTML,…", "filename": ""}], "chunk_position": 44, "heading_path": "7. Custom Filters > 7. Custom Filters", "breadcrumbs": "Fit Markdown with Pruning & BM25 > 7. Custom Filters > 7. Custom Filters"}, {"id": "669fc7ee760f3807", "url": "https://docs.crawl4ai.com/core/fit-markdown/", "page_title": "Fit Markdown with Pruning & BM25", "page_type": "guide", "page_summary": "Explains how to use Fit Markdown with Pruning and BM25 content filters in Crawl4AI to extract concise, relevant content from web pages.", "heading": "8. Final Thoughts", "content": "Page: Fit Markdown with Pruning & BM25\nSection: 8. Final Thoughts\n\n**Fit Markdown** is a crucial feature for:\n\n- **Summaries**: Quickly get the important text from a cluttered page.\n- **Search**: Combine with **BM25** to produce content relevant to a query.\n- **AI…", "code_blocks": [], "chunk_position": 44, "heading_path": "8. Final Thoughts > 8. Final Thoughts", "breadcrumbs": "Fit Markdown with Pruning & BM25 > 8. Final Thoughts > 8. Final Thoughts"}, {"id": "cce2ec7185cccf14", "url": "https://docs.crawl4ai.com/core/installation/", "page_title": "Installation & Setup (2023 Edition)", "page_type": "guide", "page_summary": "Installation and setup guide for Crawl4AI, covering basic install, diagnostics, verification, optional advanced dependencies, Docker, and local server mode.", "heading": "1. Basic Installation", "content": "Page: Installation & Setup (2023 Edition)\nSection: 1. Basic Installation\n\nThis installs the  **core**  Crawl4AI library along with essential dependencies.  **No**  advanced features (like transformers or PyTorch) are included yet.", "code_blocks": [{"language": "bash", "code": "pip install crawl4ai", "filename": ""}], "chunk_position": 45, "heading_path": "1. Basic Installation > 1. Basic Installation", "breadcrumbs": "Installation & Setup (2023 Edition) > 1. Basic Installation > 1. Basic Installation"}, {"id": "6a0c0e2cd4d29f00", "url": "https://docs.crawl4ai.com/core/installation/", "page_title": "Installation & Setup (2023 Edition)", "page_type": "guide", "page_summary": "Installation and setup guide for Crawl4AI, covering basic install, diagnostics, verification, optional advanced dependencies, Docker, and local server mode.", "heading": "2.1 Run the Setup Command", "content": "Page: Installation & Setup (2023 Edition)\nSection: 2.1 Run the Setup Command\n\nAfter installing, call:\n\n **What does it do?** \n- Installs or updates required browser dependencies for both regular and undetected modes\n- Performs OS-level checks (e.g., missing libs on Linux)\n-…", "code_blocks": [{"language": "bash", "code": "crawl4ai-setup", "filename": ""}], "chunk_position": 45, "heading_path": "2.1 Run the Setup Command > 2.1 Run the Setup Command", "breadcrumbs": "Installation & Setup (2023 Edition) > 2.1 Run the Setup Command > 2.1 Run the Setup Command"}, {"id": "708706d0be65f845", "url": "https://docs.crawl4ai.com/core/installation/", "page_title": "Installation & Setup (2023 Edition)", "page_type": "guide", "page_summary": "Installation and setup guide for Crawl4AI, covering basic install, diagnostics, verification, optional advanced dependencies, Docker, and local server mode.", "heading": "2.2 Diagnostics", "content": "Page: Installation & Setup (2023 Edition)\nSection: 2.2 Diagnostics\n\nOptionally, you can run  **diagnostics**  to confirm everything is functioning:\n\nThis command attempts to:\n- Check Python version compatibility\n- Verify Playwright installation\n- Inspect environment…", "code_blocks": [{"language": "bash", "code": "crawl4ai-doctor", "filename": ""}], "chunk_position": 45, "heading_path": "2.2 Diagnostics > 2.2 Diagnostics", "breadcrumbs": "Installation & Setup (2023 Edition) > 2.2 Diagnostics > 2.2 Diagnostics"}, {"id": "dc7c4886d63feec6", "url": "https://docs.crawl4ai.com/core/installation/", "page_title": "Installation & Setup (2023 Edition)", "page_type": "guide", "page_summary": "Installation and setup guide for Crawl4AI, covering basic install, diagnostics, verification, optional advanced dependencies, Docker, and local server mode.", "heading": "3. Verifying Installation: A Simple Crawl (Skip this step if you already run `crawl4ai-doctor`)", "content": "Page: Installation & Setup (2023 Edition)\nSection: 3. Verifying Installation: A Simple Crawl (Skip this step if you already run `crawl4ai-doctor`)\n\nBelow is a minimal Python script demonstrating a  **basic**  crawl. It uses our new  **`BrowserConfig`**  and  **`CrawlerRunConfig`**  for clarity, though no custom settings are passed in this…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig\n\nasync def main():\n    async with AsyncWebCrawler() as crawler:\n        result = await crawler.arun(…", "filename": ""}], "chunk_position": 45, "heading_path": "3. Verifying Installation: A Simple Crawl (Skip this step if you already run `crawl4ai-doctor`) > 3. Verifying Installation: A Simple Crawl (Skip this step if you already run `crawl4ai-doctor`)", "breadcrumbs": "Installation & Setup (2023 Edition) > 3. Verifying Installation: A Simple Crawl (Skip this step if you already run `crawl4ai-doctor`) > 3. Verifying Installation: A Simple Crawl (Skip this step if you already run `crawl4ai-doctor`)"}, {"id": "698f2f72d9a1f884", "url": "https://docs.crawl4ai.com/core/installation/", "page_title": "Installation & Setup (2023 Edition)", "page_type": "guide", "page_summary": "Installation and setup guide for Crawl4AI, covering basic install, diagnostics, verification, optional advanced dependencies, Docker, and local server mode.", "heading": "4. Advanced Installation (Optional)", "content": "Page: Installation & Setup (2023 Edition)\nSection: 4. Advanced Installation (Optional)\n\n**Warning** : Only install these  **if you truly need them** . They bring in larger dependencies, including big models, which can increase disk usage and memory load significantly.", "code_blocks": [], "chunk_position": 45, "heading_path": "4. Advanced Installation (Optional) > 4. Advanced Installation (Optional)", "breadcrumbs": "Installation & Setup (2023 Edition) > 4. Advanced Installation (Optional) > 4. Advanced Installation (Optional)"}, {"id": "f7ce3d26c5f7f4a8", "url": "https://docs.crawl4ai.com/core/installation/", "page_title": "Installation & Setup (2023 Edition)", "page_type": "guide", "page_summary": "Installation and setup guide for Crawl4AI, covering basic install, diagnostics, verification, optional advanced dependencies, Docker, and local server mode.", "heading": "4.1 Torch, Transformers, or All", "content": "Page: Installation & Setup (2023 Edition)\nSection: 4.1 Torch, Transformers, or All\n\n- **Text Clustering (Torch)** \n\n  Installs PyTorch-based features (e.g., cosine similarity or advanced semantic chunking).\n\n- **Transformers** \n\n  Adds Hugging Face-based summarization or generation…", "code_blocks": [{"language": "bash", "code": "pip install crawl4ai[torch]\ncrawl4ai-setup", "filename": ""}, {"language": "bash", "code": "pip install crawl4ai[transformer]\ncrawl4ai-setup", "filename": ""}, {"language": "bash", "code": "pip install crawl4ai[all]\ncrawl4ai-setup", "filename": ""}], "chunk_position": 45, "heading_path": "4.1 Torch, Transformers, or All > 4.1 Torch, Transformers, or All", "breadcrumbs": "Installation & Setup (2023 Edition) > 4.1 Torch, Transformers, or All > 4.1 Torch, Transformers, or All"}, {"id": "c0e668b71f014cc6", "url": "https://docs.crawl4ai.com/core/installation/", "page_title": "Installation & Setup (2023 Edition)", "page_type": "guide", "page_summary": "Installation and setup guide for Crawl4AI, covering basic install, diagnostics, verification, optional advanced dependencies, Docker, and local server mode.", "heading": "(Optional) Pre-Fetching Models", "content": "Page: Installation & Setup (2023 Edition)\nSection: (Optional) Pre-Fetching Models\n\nThis step caches large models locally (if needed).  **Only do this**  if your workflow requires them.", "code_blocks": [{"language": "bash", "code": "crawl4ai-download-models", "filename": ""}], "chunk_position": 45, "heading_path": "(Optional) Pre-Fetching Models > (Optional) Pre-Fetching Models", "breadcrumbs": "Installation & Setup (2023 Edition) > (Optional) Pre-Fetching Models > (Optional) Pre-Fetching Models"}, {"id": "91af46173d4b3c28", "url": "https://docs.crawl4ai.com/core/installation/", "page_title": "Installation & Setup (2023 Edition)", "page_type": "guide", "page_summary": "Installation and setup guide for Crawl4AI, covering basic install, diagnostics, verification, optional advanced dependencies, Docker, and local server mode.", "heading": "5. Docker (Experimental)", "content": "Page: Installation & Setup (2023 Edition)\nSection: 5. Docker (Experimental)\n\nWe provide a  **temporary**  Docker approach for testing.  **It’s not stable and may break**  with future releases. We plan a major Docker revamp in a future stable version, 2025 Q1. If you still…", "code_blocks": [{"language": "bash", "code": "docker pull unclecode/crawl4ai:basic\ndocker run -p 11235:11235 unclecode/crawl4ai:basic", "filename": ""}], "chunk_position": 45, "heading_path": "5. Docker (Experimental) > 5. Docker (Experimental)", "breadcrumbs": "Installation & Setup (2023 Edition) > 5. Docker (Experimental) > 5. Docker (Experimental)"}, {"id": "53d8581b758c7ee8", "url": "https://docs.crawl4ai.com/core/installation/", "page_title": "Installation & Setup (2023 Edition)", "page_type": "guide", "page_summary": "Installation and setup guide for Crawl4AI, covering basic install, diagnostics, verification, optional advanced dependencies, Docker, and local server mode.", "heading": "6. Local Server Mode (Legacy)", "content": "Page: Installation & Setup (2023 Edition)\nSection: 6. Local Server Mode (Legacy)\n\nSome older docs mention running Crawl4AI as a local server. This approach has been  **partially replaced**  by the new Docker-based prototype and upcoming stable server release. You can experiment,…", "code_blocks": [], "chunk_position": 45, "heading_path": "6. Local Server Mode (Legacy) > 6. Local Server Mode (Legacy)", "breadcrumbs": "Installation & Setup (2023 Edition) > 6. Local Server Mode (Legacy) > 6. Local Server Mode (Legacy)"}, {"id": "9346a6d3c54394e7", "url": "https://docs.crawl4ai.com/core/installation/", "page_title": "Installation & Setup (2023 Edition)", "page_type": "guide", "page_summary": "Installation and setup guide for Crawl4AI, covering basic install, diagnostics, verification, optional advanced dependencies, Docker, and local server mode.", "heading": "Summary", "content": "Page: Installation & Setup (2023 Edition)\nSection: Summary\n\n1. **Install** with `pip install crawl4ai` and run `crawl4ai-setup`.\n2. **Diagnose** with `crawl4ai-doctor` if you see errors.\n3. **Verify** by crawling `example.com` with minimal `BrowserConfig` +…", "code_blocks": [], "chunk_position": 45, "heading_path": "Summary > Summary", "breadcrumbs": "Installation & Setup (2023 Edition) > Summary > Summary"}, {"id": "eda6efd564c93dbd", "url": "https://docs.crawl4ai.com/core/link-media/", "page_title": "Link & Media - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial covers how to extract and filter links (internal/external) and media (images, videos, audio) from crawled pages using Crawl4AI, including advanced link head extraction with scoring,…", "heading": "1. Link Extraction", "content": "Page: Link & Media - Crawl4AI Documentation (v0.9.x)\nSection: 1. Link Extraction\n\nWhen you call `arun()` or `arun_many()` on a URL, Crawl4AI automatically extracts links and stores them in the `links` field of `CrawlResult`. By default, the crawler tries to distinguish…", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler\n\nasync with AsyncWebCrawler() as crawler:\n    result = await crawler.arun(\"https://www.example.com\")\n    if result.success:\n        internal_links =…", "filename": ""}, {"language": "python", "code": "result.links = {\n  \"internal\": [\n    {\n      \"href\": \"https://kidocode.com/\",\n      \"text\": \"\",\n      \"title\": \"\",\n      \"base_domain\": \"kidocode.com\"\n    },\n    {\n      \"href\":…", "filename": ""}], "chunk_position": 46, "heading_path": "1. Link Extraction > 1. Link Extraction", "breadcrumbs": "Link & Media - Crawl4AI Documentation (v0.9.x) > 1. Link Extraction > 1. Link Extraction"}, {"id": "315b2eeef19cd720", "url": "https://docs.crawl4ai.com/core/link-media/", "page_title": "Link & Media - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial covers how to extract and filter links (internal/external) and media (images, videos, audio) from crawled pages using Crawl4AI, including advanced link head extraction with scoring,…", "heading": "2. Advanced Link Head Extraction & Scoring", "content": "Page: Link & Media - Crawl4AI Documentation (v0.9.x)\nSection: 2. Advanced Link Head Extraction & Scoring\n\nEver wanted to not just extract links, but also get the actual content (title, description, metadata) from those linked pages? And score them for relevance? This is exactly what Link Head Extraction…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai import LinkPreviewConfig\n\nasync def extract_link_heads_example():\n    \"\"\"\n    Complete example showing link head…", "filename": ""}, {"language": "text", "code": "✅ Successfully crawled: https://docs.python.org/3/\n📄 Page title: 3.13.5 Documentation\n🔗 Found 53 internal links\n🌍 Found 1 external links\n🧠 Links with head data extracted: 10\n\n🏆 Top 3 Links with Full…", "filename": ""}, {"language": "python", "code": "from crawl4ai import LinkPreviewConfig\n\nlink_preview_config = LinkPreviewConfig(\n    # BASIC SETTINGS\n    verbose=True,                    # Show detailed logs (recommended for learning)\n\n    # LINK…", "filename": ""}, {"language": "python", "code": "# High intrinsic score indicators:\n# ✅ Clean URL structure (docs.python.org/api/reference)\n# ✅ Meaningful link text (\"API Reference Guide\")\n# ✅ Relevant to page context\n# ✅ Not buried deep in…", "filename": ""}, {"language": "python", "code": "# Example: query = \"machine learning tutorial\"\n# High contextual score: Link to \"Complete Machine Learning Guide\"\n# Low contextual score: Link to \"Privacy Policy\"", "filename": ""}, {"language": "python", "code": "# When both scores available: (intrinsic * 0.3) + (contextual * 0.7)\n# When only intrinsic: uses intrinsic score\n# When only contextual: uses contextual score\n# When neither: not calculated", "filename": ""}, {"language": "python", "code": "async def research_assistant():\n    config = CrawlerRunConfig(\n        link_preview_config=LinkPreviewConfig(\n            include_internal=True,\n            include_external=True,…", "filename": ""}, {"language": "python", "code": "async def api_discovery():\n    config = CrawlerRunConfig(\n        link_preview_config=LinkPreviewConfig(\n            include_internal=True,\n            include_patterns=[\"*/api/*\", \"*/reference/*\"],…", "filename": ""}, {"language": "python", "code": "async def quality_analysis():\n    config = CrawlerRunConfig(\n        link_preview_config=LinkPreviewConfig(\n            include_internal=True,\n            max_links=200,\n            concurrency=20,…", "filename": ""}, {"language": "python", "code": "# Check your configuration:\nconfig = CrawlerRunConfig(\n    link_preview_config=LinkPreviewConfig(\n        verbose=True   # ← Enable to see what's happening\n    )\n)", "filename": ""}, {"language": "python", "code": "# Make sure scoring is enabled:\nconfig = CrawlerRunConfig(\n    score_links=True,  # ← Enable intrinsic scoring\n    link_preview_config=LinkPreviewConfig(\n        query=\"your search terms\"  # ← For…", "filename": ""}, {"language": "python", "code": "# Optimize performance:\nlink_preview_config = LinkPreviewConfig(\n    max_links=20,      # ← Reduce number\n    concurrency=10,    # ← Increase parallelism\n    timeout=3,         # ← Shorter timeout…", "filename": ""}], "chunk_position": 46, "heading_path": "2. Advanced Link Head Extraction & Scoring > 2. Advanced Link Head Extraction & Scoring", "breadcrumbs": "Link & Media - Crawl4AI Documentation (v0.9.x) > 2. Advanced Link Head Extraction & Scoring > 2. Advanced Link Head Extraction & Scoring"}, {"id": "771a8e8298524138", "url": "https://docs.crawl4ai.com/core/link-media/", "page_title": "Link & Media - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial covers how to extract and filter links (internal/external) and media (images, videos, audio) from crawled pages using Crawl4AI, including advanced link head extraction with scoring,…", "heading": "3. Domain Filtering", "content": "Page: Link & Media - Crawl4AI Documentation (v0.9.x)\nSection: 3. Domain Filtering\n\nSome websites contain hundreds of third-party or affiliate links. You can filter out certain domains at **crawl time** by configuring the crawler. The most relevant parameters in `CrawlerRunConfig`…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig\n\nasync def main():\n    crawler_cfg = CrawlerRunConfig(\n        exclude_external_links=True,          # No links…", "filename": ""}, {"language": "python", "code": "crawler_cfg = CrawlerRunConfig(\n    exclude_domains=[\"suspiciousads.com\"]\n)", "filename": ""}], "chunk_position": 46, "heading_path": "3. Domain Filtering > 3. Domain Filtering", "breadcrumbs": "Link & Media - Crawl4AI Documentation (v0.9.x) > 3. Domain Filtering > 3. Domain Filtering"}, {"id": "a2317950412c684b", "url": "https://docs.crawl4ai.com/core/link-media/", "page_title": "Link & Media - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial covers how to extract and filter links (internal/external) and media (images, videos, audio) from crawled pages using Crawl4AI, including advanced link head extraction with scoring,…", "heading": "4. Media Extraction", "content": "Page: Link & Media - Crawl4AI Documentation (v0.9.x)\nSection: 4. Media Extraction\n\n###### 4.1 Accessing `result.media`\n\nBy default, Crawl4AI collects images, audio and video URLs it finds on the page. These are stored in `result.media`, a dictionary keyed by media type (e.g.,…", "code_blocks": [{"language": "python", "code": "if result.success:\n    # Get images\n    images_info = result.media.get(\"images\", [])\n    print(f\"Found {len(images_info)} images in total.\")\n    for i, img in enumerate(images_info[:3]):  # Inspect…", "filename": ""}, {"language": "python", "code": "result.media = {\n  \"images\": [\n    {\n      \"src\": \"https://cdn.prod.website-files.com/.../Group%2089.svg\",\n      \"alt\": \"coding school for kids\",\n      \"desc\": \"Trial Class Degrees degrees All…", "filename": ""}, {"language": "python", "code": "crawler_cfg = CrawlerRunConfig(\n    exclude_external_images=True\n)", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nasync def main():\n    crawler_cfg = CrawlerRunConfig(\n        capture_mhtml=True  # Enable MHTML capture\n    )\n\n    async with…", "filename": ""}], "chunk_position": 46, "heading_path": "4. Media Extraction > 4. Media Extraction", "breadcrumbs": "Link & Media - Crawl4AI Documentation (v0.9.x) > 4. Media Extraction > 4. Media Extraction"}, {"id": "bd1dbd8145582132", "url": "https://docs.crawl4ai.com/core/link-media/", "page_title": "Link & Media - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial covers how to extract and filter links (internal/external) and media (images, videos, audio) from crawled pages using Crawl4AI, including advanced link head extraction with scoring,…", "heading": "5. Putting It All Together: Link & Media Filtering", "content": "Page: Link & Media - Crawl4AI Documentation (v0.9.x)\nSection: 5. Putting It All Together: Link & Media Filtering\n\nHere’s a combined example demonstrating how to filter out external links, skip certain domains, and exclude external images:", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig\n\nasync def main():\n    # Suppose we want to keep only internal links, remove certain domains, \n    # and discard…", "filename": ""}], "chunk_position": 46, "heading_path": "5. Putting It All Together: Link & Media Filtering > 5. Putting It All Together: Link & Media Filtering", "breadcrumbs": "Link & Media - Crawl4AI Documentation (v0.9.x) > 5. Putting It All Together: Link & Media Filtering > 5. Putting It All Together: Link & Media Filtering"}, {"id": "fed80a61d3a5e821", "url": "https://docs.crawl4ai.com/core/link-media/", "page_title": "Link & Media - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This tutorial covers how to extract and filter links (internal/external) and media (images, videos, audio) from crawled pages using Crawl4AI, including advanced link head extraction with scoring,…", "heading": "6. Common Pitfalls & Tips", "content": "Page: Link & Media - Crawl4AI Documentation (v0.9.x)\nSection: 6. Common Pitfalls & Tips\n\n1. **Conflicting Flags**:\n   - `exclude_external_links=True` but then also specifying `exclude_social_media_links=True` is typically fine, but understand that the first setting already discards *all*…", "code_blocks": [], "chunk_position": 46, "heading_path": "6. Common Pitfalls & Tips > 6. Common Pitfalls & Tips", "breadcrumbs": "Link & Media - Crawl4AI Documentation (v0.9.x) > 6. Common Pitfalls & Tips > 6. Common Pitfalls & Tips"}, {"id": "9d3739d4a2e353bd", "url": "https://docs.crawl4ai.com/core/llmtxt/", "page_title": "Llmtxt - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page documents the Llmtxt feature in Crawl4AI, which provides LLM-friendly text for crawled content. The provided raw content contains only navigation and UI elements, so the actual…", "heading": "Llmtxt", "content": "Page: Llmtxt - Crawl4AI Documentation (v0.9.x)\nSection: Llmtxt\n\nThe provided raw content for this page contains only navigation and UI elements. No actual documentation text was found in the provided content.", "code_blocks": [], "chunk_position": 47, "heading_path": "Llmtxt > Llmtxt", "breadcrumbs": "Llmtxt - Crawl4AI Documentation (v0.9.x) > Llmtxt > Llmtxt"}, {"id": "07ccfb0727a4e915", "url": "https://docs.crawl4ai.com/core/local-files/", "page_title": "Prefix-Based Input Handling in Crawl4AI", "page_type": "guide", "page_summary": "This guide demonstrates how to use Crawl4AI to crawl web URLs, local HTML files, and raw HTML strings using prefix-based input handling with the unified `url` parameter and `CrawlerRunConfig`.", "heading": "Crawling a Web URL", "content": "Page: Prefix-Based Input Handling in Crawl4AI\nSection: Crawling a Web URL\n\nTo crawl a live web page, provide the URL starting with `http://` or `https://`, using a `CrawlerRunConfig` object:", "code_blocks": [{"language": "", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CacheMode, CrawlerRunConfig\n\nasync def crawl_web():\n    config = CrawlerRunConfig(cache_mode=CacheMode.BYPASS)\n    async with AsyncWebCrawler() as…", "filename": ""}], "chunk_position": 48, "heading_path": "Crawling a Web URL > Crawling a Web URL", "breadcrumbs": "Prefix-Based Input Handling in Crawl4AI > Crawling a Web URL > Crawling a Web URL"}, {"id": "90697646b968a920", "url": "https://docs.crawl4ai.com/core/local-files/", "page_title": "Prefix-Based Input Handling in Crawl4AI", "page_type": "guide", "page_summary": "This guide demonstrates how to use Crawl4AI to crawl web URLs, local HTML files, and raw HTML strings using prefix-based input handling with the unified `url` parameter and `CrawlerRunConfig`.", "heading": "Crawling a Local HTML File", "content": "Page: Prefix-Based Input Handling in Crawl4AI\nSection: Crawling a Local HTML File\n\nTo crawl a local HTML file, prefix the file path with `file://`.", "code_blocks": [{"language": "", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CacheMode, CrawlerRunConfig\n\nasync def crawl_local_file():\n    local_file_path = \"/path/to/apple.html\"  # Replace with your file path\n    file_url…", "filename": ""}], "chunk_position": 48, "heading_path": "Crawling a Local HTML File > Crawling a Local HTML File", "breadcrumbs": "Prefix-Based Input Handling in Crawl4AI > Crawling a Local HTML File > Crawling a Local HTML File"}, {"id": "b1aca72f5431de1b", "url": "https://docs.crawl4ai.com/core/local-files/", "page_title": "Prefix-Based Input Handling in Crawl4AI", "page_type": "guide", "page_summary": "This guide demonstrates how to use Crawl4AI to crawl web URLs, local HTML files, and raw HTML strings using prefix-based input handling with the unified `url` parameter and `CrawlerRunConfig`.", "heading": "Crawling Raw HTML Content", "content": "Page: Prefix-Based Input Handling in Crawl4AI\nSection: Crawling Raw HTML Content\n\nTo crawl raw HTML content, prefix the HTML string with `raw:`.", "code_blocks": [{"language": "", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CacheMode\nfrom crawl4ai.async_configs import CrawlerRunConfig\n\nasync def crawl_raw_html():\n    raw_html = \"<html><body><h1>Hello,…", "filename": ""}], "chunk_position": 48, "heading_path": "Crawling Raw HTML Content > Crawling Raw HTML Content", "breadcrumbs": "Prefix-Based Input Handling in Crawl4AI > Crawling Raw HTML Content > Crawling Raw HTML Content"}, {"id": "37e43f106ac2ba6b", "url": "https://docs.crawl4ai.com/core/local-files/", "page_title": "Prefix-Based Input Handling in Crawl4AI", "page_type": "guide", "page_summary": "This guide demonstrates how to use Crawl4AI to crawl web URLs, local HTML files, and raw HTML strings using prefix-based input handling with the unified `url` parameter and `CrawlerRunConfig`.", "heading": "Complete Example", "content": "Page: Prefix-Based Input Handling in Crawl4AI\nSection: Complete Example\n\nBelow is a comprehensive script that:\n\n- Crawls the Wikipedia page for \"Apple.\"\n- Saves the HTML content to a local file (`apple.html`).\n- Crawls the local HTML file and verifies the markdown length…", "code_blocks": [{"language": "", "code": "import os\nimport sys\nimport asyncio\nfrom pathlib import Path\nfrom crawl4ai import AsyncWebCrawler, CacheMode, CrawlerRunConfig\n\nasync def main():\n    wikipedia_url =…", "filename": ""}], "chunk_position": 48, "heading_path": "Complete Example > Complete Example", "breadcrumbs": "Prefix-Based Input Handling in Crawl4AI > Complete Example > Complete Example"}, {"id": "92ca6349ed061c2b", "url": "https://docs.crawl4ai.com/core/local-files/", "page_title": "Prefix-Based Input Handling in Crawl4AI", "page_type": "guide", "page_summary": "This guide demonstrates how to use Crawl4AI to crawl web URLs, local HTML files, and raw HTML strings using prefix-based input handling with the unified `url` parameter and `CrawlerRunConfig`.", "heading": "Conclusion", "content": "Page: Prefix-Based Input Handling in Crawl4AI\nSection: Conclusion\n\nWith the unified `url` parameter and prefix-based handling in **Crawl4AI** , you can seamlessly handle web URLs, local HTML files, and raw HTML content. Use `CrawlerRunConfig` for flexible and…", "code_blocks": [], "chunk_position": 48, "heading_path": "Conclusion > Conclusion", "breadcrumbs": "Prefix-Based Input Handling in Crawl4AI > Conclusion > Conclusion"}, {"id": "96b0075e47370da9", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "Prerequisites", "content": "Page: Markdown Generation Basics\nSection: Prerequisites\n\n> **Prerequisites** \n> \n> - You’ve completed or read [AsyncWebCrawler Basics](../simple-crawling/) to understand how to run a simple crawl.\n> \n> - You know how to configure `CrawlerRunConfig`.", "code_blocks": [], "chunk_position": 49, "heading_path": "Prerequisites > Prerequisites", "breadcrumbs": "Markdown Generation Basics > Prerequisites > Prerequisites"}, {"id": "55d2d92ffaab9c50", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "1. Quick Example", "content": "Page: Markdown Generation Basics\nSection: 1. Quick Example\n\nHere’s a minimal code snippet that uses the **DefaultMarkdownGenerator** with no additional filtering:\n\n**What’s happening?** \n\n- `CrawlerRunConfig( markdown_generator = DefaultMarkdownGenerator() )`…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai.markdown_generation_strategy import DefaultMarkdownGenerator\n\nasync def main():\n    config = CrawlerRunConfig(…", "filename": ""}], "chunk_position": 49, "heading_path": "1. Quick Example > 1. Quick Example", "breadcrumbs": "Markdown Generation Basics > 1. Quick Example > 1. Quick Example"}, {"id": "60fa250986ef5efe", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "2. How Markdown Generation Works", "content": "Page: Markdown Generation Basics\nSection: 2. How Markdown Generation Works\n\nUnder the hood, **DefaultMarkdownGenerator** uses a specialized HTML-to-text approach that:\n\n- Preserves headings, code blocks, bullet points, etc.\n- Removes extraneous tags (scripts, styles) that…", "code_blocks": [], "chunk_position": 49, "heading_path": "2. How Markdown Generation Works > 2. How Markdown Generation Works", "breadcrumbs": "Markdown Generation Basics > 2. How Markdown Generation Works > 2. How Markdown Generation Works"}, {"id": "686b333aa947afc1", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "2.1 HTML-to-Text Conversion (Forked & Modified)", "content": "Page: Markdown Generation Basics\nSection: 2.1 HTML-to-Text Conversion (Forked & Modified)\n\nUnder the hood, **DefaultMarkdownGenerator** uses a specialized HTML-to-text approach that:\n\n- Preserves headings, code blocks, bullet points, etc.\n- Removes extraneous tags (scripts, styles) that…", "code_blocks": [], "chunk_position": 49, "heading_path": "2.1 HTML-to-Text Conversion (Forked & Modified) > 2.1 HTML-to-Text Conversion (Forked & Modified)", "breadcrumbs": "Markdown Generation Basics > 2.1 HTML-to-Text Conversion (Forked & Modified) > 2.1 HTML-to-Text Conversion (Forked & Modified)"}, {"id": "3d3db21a2cc36a0d", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "2.2 Link Citations & References", "content": "Page: Markdown Generation Basics\nSection: 2.2 Link Citations & References\n\nBy default, the generator can convert `<a href=\"...\">` elements into `[text][1]` citations, then place the actual links at the bottom of the document. This is handy for research workflows that demand…", "code_blocks": [], "chunk_position": 49, "heading_path": "2.2 Link Citations & References > 2.2 Link Citations & References", "breadcrumbs": "Markdown Generation Basics > 2.2 Link Citations & References > 2.2 Link Citations & References"}, {"id": "90e29091bc6b1c9f", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "2.3 Optional Content Filters", "content": "Page: Markdown Generation Basics\nSection: 2.3 Optional Content Filters\n\nBefore or after the HTML-to-Markdown step, you can apply a **content filter** (like BM25 or Pruning) to reduce noise and produce a “fit_markdown”—a heavily pruned version focusing on the page’s main…", "code_blocks": [], "chunk_position": 49, "heading_path": "2.3 Optional Content Filters > 2.3 Optional Content Filters", "breadcrumbs": "Markdown Generation Basics > 2.3 Optional Content Filters > 2.3 Optional Content Filters"}, {"id": "02af3faaec5563e6", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "3. Configuring the Default Markdown Generator", "content": "Page: Markdown Generation Basics\nSection: 3. Configuring the Default Markdown Generator\n\nYou can tweak the output by passing an `options` dict to `DefaultMarkdownGenerator`. For example:\n\nSome commonly used `options`:\n\n- **`ignore_links`** (bool): Whether to remove all hyperlinks in the…", "code_blocks": [{"language": "python", "code": "from crawl4ai.markdown_generation_strategy import DefaultMarkdownGenerator\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nasync def main():\n    # Example: ignore all links, don't escape…", "filename": ""}], "chunk_position": 49, "heading_path": "3. Configuring the Default Markdown Generator > 3. Configuring the Default Markdown Generator", "breadcrumbs": "Markdown Generation Basics > 3. Configuring the Default Markdown Generator > 3. Configuring the Default Markdown Generator"}, {"id": "c58ba9f967d3f396", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "4. Selecting the HTML Source for Markdown Generation", "content": "Page: Markdown Generation Basics\nSection: 4. Selecting the HTML Source for Markdown Generation\n\nThe `content_source` parameter allows you to control which HTML content is used as input for markdown generation. This gives you flexibility in how the HTML is processed before conversion to markdown.", "code_blocks": [{"language": "python", "code": "from crawl4ai.markdown_generation_strategy import DefaultMarkdownGenerator\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nasync def main():\n    # Option 1: Use the raw HTML directly from the…", "filename": ""}], "chunk_position": 49, "heading_path": "4. Selecting the HTML Source for Markdown Generation > 4. Selecting the HTML Source for Markdown Generation", "breadcrumbs": "Markdown Generation Basics > 4. Selecting the HTML Source for Markdown Generation > 4. Selecting the HTML Source for Markdown Generation"}, {"id": "7f6001a8823503ac", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "HTML Source Options", "content": "Page: Markdown Generation Basics\nSection: HTML Source Options\n\n- **`\"cleaned_html\"`** (default): Uses the HTML after it has been processed by the scraping strategy. This HTML is typically cleaner and more focused on content, with some boilerplate removed.\n-…", "code_blocks": [], "chunk_position": 49, "heading_path": "HTML Source Options > HTML Source Options", "breadcrumbs": "Markdown Generation Basics > HTML Source Options > HTML Source Options"}, {"id": "fccab4eccd772151", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "When to Use Each Option", "content": "Page: Markdown Generation Basics\nSection: When to Use Each Option\n\n- Use **`\"cleaned_html\"`** (default) for most cases where you want a balance of content preservation and noise removal.\n- Use **`\"raw_html\"`** when you need to preserve all original content, or when…", "code_blocks": [], "chunk_position": 49, "heading_path": "When to Use Each Option > When to Use Each Option", "breadcrumbs": "Markdown Generation Basics > When to Use Each Option > When to Use Each Option"}, {"id": "767fb43087bf2189", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "5. Content Filters", "content": "Page: Markdown Generation Basics\nSection: 5. Content Filters\n\n**Content filters** selectively remove or rank sections of text before turning them into Markdown. This is especially helpful if your page has ads, nav bars, or other clutter you don’t want.", "code_blocks": [], "chunk_position": 49, "heading_path": "5. Content Filters > 5. Content Filters", "breadcrumbs": "Markdown Generation Basics > 5. Content Filters > 5. Content Filters"}, {"id": "6b59d8ed62b589b6", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "5.1 BM25ContentFilter", "content": "Page: Markdown Generation Basics\nSection: 5.1 BM25ContentFilter\n\nIf you have a **search query**, BM25 is a good choice:\n\n- **`user_query`**: The term you want to focus on. BM25 tries to keep only content blocks relevant to that query.\n- **`bm25_threshold`**: Raise…", "code_blocks": [{"language": "python", "code": "from crawl4ai.markdown_generation_strategy import DefaultMarkdownGenerator\nfrom crawl4ai.content_filter_strategy import BM25ContentFilter\nfrom crawl4ai import CrawlerRunConfig\n\nbm25_filter =…", "filename": ""}], "chunk_position": 49, "heading_path": "5.1 BM25ContentFilter > 5.1 BM25ContentFilter", "breadcrumbs": "Markdown Generation Basics > 5.1 BM25ContentFilter > 5.1 BM25ContentFilter"}, {"id": "7b07b97930c33f18", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "5.2 PruningContentFilter", "content": "Page: Markdown Generation Basics\nSection: 5.2 PruningContentFilter\n\nIf you **don’t** have a specific query, or if you just want a robust “junk remover,” use `PruningContentFilter`. It analyzes text density, link density, HTML structure, and known patterns (like…", "code_blocks": [{"language": "python", "code": "from crawl4ai.content_filter_strategy import PruningContentFilter\n\nprune_filter = PruningContentFilter(\n    threshold=0.5,\n    threshold_type=\"fixed\",  # or \"dynamic\"\n    min_word_threshold=50\n)", "filename": ""}], "chunk_position": 49, "heading_path": "5.2 PruningContentFilter > 5.2 PruningContentFilter", "breadcrumbs": "Markdown Generation Basics > 5.2 PruningContentFilter > 5.2 PruningContentFilter"}, {"id": "da1eec29ed22a2d7", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "5.3 LLMContentFilter", "content": "Page: Markdown Generation Basics\nSection: 5.3 LLMContentFilter\n\nFor intelligent content filtering and high-quality markdown generation, you can use the **LLMContentFilter**. This filter leverages LLMs to generate relevant markdown while preserving the original…", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, LLMConfig, DefaultMarkdownGenerator\nfrom crawl4ai.content_filter_strategy import LLMContentFilter\n\nasync def main():\n    #…", "filename": ""}, {"language": "python", "code": "filter = LLMContentFilter(\n    instruction=\"\"\"\n    Extract the main educational content while preserving its original wording and substance completely.\n    1. Maintain the exact language and…", "filename": ""}, {"language": "python", "code": "filter = LLMContentFilter(\n    instruction=\"\"\"\n    Focus on extracting specific types of content:\n    - Technical documentation\n    - Code examples\n    - API references\n    Reformat the content into…", "filename": ""}], "chunk_position": 49, "heading_path": "5.3 LLMContentFilter > 5.3 LLMContentFilter", "breadcrumbs": "Markdown Generation Basics > 5.3 LLMContentFilter > 5.3 LLMContentFilter"}, {"id": "7d18f68f1fa5c949", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "6. Using Fit Markdown", "content": "Page: Markdown Generation Basics\nSection: 6. Using Fit Markdown\n\nWhen a content filter is active, the library produces two forms of markdown inside `result.markdown`:\n\n1. **`raw_markdown`**: The full unfiltered markdown.\n2. **`fit_markdown`**: A “fit” version…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai.markdown_generation_strategy import DefaultMarkdownGenerator\nfrom crawl4ai.content_filter_strategy import…", "filename": ""}], "chunk_position": 49, "heading_path": "6. Using Fit Markdown > 6. Using Fit Markdown", "breadcrumbs": "Markdown Generation Basics > 6. Using Fit Markdown > 6. Using Fit Markdown"}, {"id": "298c37f020aa4d83", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "7. The `MarkdownGenerationResult` Object", "content": "Page: Markdown Generation Basics\nSection: 7. The `MarkdownGenerationResult` Object\n\nIf your library stores detailed markdown output in an object like `MarkdownGenerationResult`, you’ll see fields such as:\n\n- **`raw_markdown`**: The direct HTML-to-markdown transformation (no…", "code_blocks": [{"language": "python", "code": "md_obj = result.markdown  # your library’s naming may vary\nprint(\"RAW:\\n\", md_obj.raw_markdown)\nprint(\"CITED:\\n\", md_obj.markdown_with_citations)\nprint(\"REFERENCES:\\n\",…", "filename": ""}], "chunk_position": 49, "heading_path": "7. The `MarkdownGenerationResult` Object > 7. The `MarkdownGenerationResult` Object", "breadcrumbs": "Markdown Generation Basics > 7. The `MarkdownGenerationResult` Object > 7. The `MarkdownGenerationResult` Object"}, {"id": "c40a7f774d9e9855", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "8. Combining Filters (BM25 + Pruning) in Two Passes", "content": "Page: Markdown Generation Basics\nSection: 8. Combining Filters (BM25 + Pruning) in Two Passes\n\nYou might want to **prune out** noisy boilerplate first (with `PruningContentFilter`), and then **rank what’s left** against a user query (with `BM25ContentFilter`). You don’t have to crawl the page…", "code_blocks": [], "chunk_position": 49, "heading_path": "8. Combining Filters (BM25 + Pruning) in Two Passes > 8. Combining Filters (BM25 + Pruning) in Two Passes", "breadcrumbs": "Markdown Generation Basics > 8. Combining Filters (BM25 + Pruning) in Two Passes > 8. Combining Filters (BM25 + Pruning) in Two Passes"}, {"id": "3e008427bb1256cc", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "Two-Pass Example", "content": "Page: Markdown Generation Basics\nSection: Two-Pass Example\n\n###### What’s Happening?\n\n1. **Raw HTML**: We crawl once and store the raw HTML in `result.html`.\n2. **PruningContentFilter**: Takes HTML + optional parameters. It extracts blocks of text or partial…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai.content_filter_strategy import PruningContentFilter, BM25ContentFilter\nfrom bs4 import BeautifulSoup\n\nasync def…", "filename": ""}], "chunk_position": 49, "heading_path": "Two-Pass Example > Two-Pass Example", "breadcrumbs": "Markdown Generation Basics > Two-Pass Example > Two-Pass Example"}, {"id": "8b782da5e878fdd4", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "Tips & Variations", "content": "Page: Markdown Generation Basics\nSection: Tips & Variations\n\n- **Plain Text vs. HTML**: If your pruned output is mostly text, BM25 can still handle it; just keep in mind it expects a valid string input. If you supply partial HTML (like `\"<p>some text</p>\"`),…", "code_blocks": [], "chunk_position": 49, "heading_path": "Tips & Variations > Tips & Variations", "breadcrumbs": "Markdown Generation Basics > Tips & Variations > Tips & Variations"}, {"id": "70a72fdf794f0789", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "One-Pass Combination?", "content": "Page: Markdown Generation Basics\nSection: One-Pass Combination?\n\nIf your codebase or pipeline design allows applying multiple filters in one pass, you could do so. But often it’s simpler—and more transparent—to run them sequentially, analyzing each step’s…", "code_blocks": [], "chunk_position": 49, "heading_path": "One-Pass Combination? > One-Pass Combination?", "breadcrumbs": "Markdown Generation Basics > One-Pass Combination? > One-Pass Combination?"}, {"id": "83c9601bb33b5c42", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "9. Common Pitfalls & Tips", "content": "Page: Markdown Generation Basics\nSection: 9. Common Pitfalls & Tips\n\n1. **No Markdown Output?** \n   - Make sure the crawler actually retrieved HTML. If the site is heavily JS-based, you may need to enable dynamic rendering or wait for elements.\n   - Check if your…", "code_blocks": [], "chunk_position": 49, "heading_path": "9. Common Pitfalls & Tips > 9. Common Pitfalls & Tips", "breadcrumbs": "Markdown Generation Basics > 9. Common Pitfalls & Tips > 9. Common Pitfalls & Tips"}, {"id": "a9e5d21b1cbb11f2", "url": "https://docs.crawl4ai.com/core/markdown-generation/", "page_title": "Markdown Generation Basics", "page_type": "guide", "page_summary": "This tutorial explains how to generate clean, structured markdown from web pages using Crawl4AI's DefaultMarkdownGenerator, including configuration options, content filters (BM25, Pruning, LLM), and…", "heading": "10. Summary & Next Steps", "content": "Page: Markdown Generation Basics\nSection: 10. Summary & Next Steps\n\nIn this **Markdown Generation Basics** tutorial, you learned to:\n\n- Configure the **DefaultMarkdownGenerator** with HTML-to-text options.\n- Select different HTML sources using the `content_source`…", "code_blocks": [], "chunk_position": 49, "heading_path": "10. Summary & Next Steps > 10. Summary & Next Steps", "breadcrumbs": "Markdown Generation Basics > 10. Summary & Next Steps > 10. Summary & Next Steps"}, {"id": "3ea327141a22bfd3", "url": "https://docs.crawl4ai.com/core/page-interaction/", "page_title": "Page Interaction - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains how to interact with dynamic webpages using Crawl4AI, covering JavaScript execution, wait conditions, multi-step flows, Shadow DOM flattening, and virtual scrolling.", "heading": "Page Interaction", "content": "Page: Page Interaction - Crawl4AI Documentation (v0.9.x)\nSection: Page Interaction\n\nCrawl4AI provides powerful features for interacting with **dynamic** webpages, handling JavaScript execution, waiting for conditions, and managing multi-step flows. By combining **js_code**,…", "code_blocks": [], "chunk_position": 50, "heading_path": "Page Interaction > Page Interaction", "breadcrumbs": "Page Interaction - Crawl4AI Documentation (v0.9.x) > Page Interaction > Page Interaction"}, {"id": "eb4ae1727e924448", "url": "https://docs.crawl4ai.com/core/page-interaction/", "page_title": "Page Interaction - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains how to interact with dynamic webpages using Crawl4AI, covering JavaScript execution, wait conditions, multi-step flows, Shadow DOM flattening, and virtual scrolling.", "heading": "1. JavaScript Execution", "content": "Page: Page Interaction - Crawl4AI Documentation (v0.9.x)\nSection: 1. JavaScript Execution\n\n###### Basic Execution\n\n**`js_code`** in **`CrawlerRunConfig`** accepts either a single JS string or a list of JS snippets. It runs **after** `wait_for` and `delay_before_return_html` — so the page is…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nasync def main():\n    # Single JS command\n    config = CrawlerRunConfig(\n        js_code=\"window.scrollTo(0,…", "filename": ""}, {"language": "text", "code": "1. Page navigation (page.goto)\n2. js_code_before_wait     ← triggers loading / clicks tabs\n3. wait_for                ← waits for content to appear\n4. delay_before_return_html ← extra safety…", "filename": ""}, {"language": "python", "code": "config = CrawlerRunConfig(\n    # Click a tab first\n    js_code_before_wait=\"document.querySelector('#specs-tab')?.click();\",\n    # Then wait for the tab content to appear…", "filename": ""}], "chunk_position": 50, "heading_path": "1. JavaScript Execution > 1. JavaScript Execution", "breadcrumbs": "Page Interaction - Crawl4AI Documentation (v0.9.x) > 1. JavaScript Execution > 1. JavaScript Execution"}, {"id": "eba45fab7c2152de", "url": "https://docs.crawl4ai.com/core/page-interaction/", "page_title": "Page Interaction - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains how to interact with dynamic webpages using Crawl4AI, covering JavaScript execution, wait conditions, multi-step flows, Shadow DOM flattening, and virtual scrolling.", "heading": "2. Wait Conditions", "content": "Page: Page Interaction - Crawl4AI Documentation (v0.9.x)\nSection: 2. Wait Conditions\n\n###### 2.1 CSS-Based Waiting\n\nSometimes, you just want to wait for a specific element to appear. For example:\n\n**Key param**:\n- **`wait_for=\"css:...\"`**: Tells the crawler to wait until that CSS…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nasync def main():\n    config = CrawlerRunConfig(\n        # Wait for at least 30 items on Hacker News…", "filename": ""}, {"language": "python", "code": "wait_condition = \"\"\"() => {\n    const items = document.querySelectorAll('.athing');\n    return items.length > 50;  // Wait for at least 51 items\n}\"\"\"\n\nconfig =…", "filename": ""}], "chunk_position": 50, "heading_path": "2. Wait Conditions > 2. Wait Conditions", "breadcrumbs": "Page Interaction - Crawl4AI Documentation (v0.9.x) > 2. Wait Conditions > 2. Wait Conditions"}, {"id": "cd14001780a1d5e3", "url": "https://docs.crawl4ai.com/core/page-interaction/", "page_title": "Page Interaction - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains how to interact with dynamic webpages using Crawl4AI, covering JavaScript execution, wait conditions, multi-step flows, Shadow DOM flattening, and virtual scrolling.", "heading": "3. Handling Dynamic Content", "content": "Page: Page Interaction - Crawl4AI Documentation (v0.9.x)\nSection: 3. Handling Dynamic Content\n\nMany modern sites require **multiple steps**: scrolling, clicking “Load More,” or updating via JavaScript. Below are typical patterns.\n\n###### 3.1 Load More Example (Hacker News “More” Link)\n\n**Key…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nasync def main():\n    # Step 1: Load initial Hacker News page\n    config = CrawlerRunConfig(…", "filename": ""}, {"language": "python", "code": "js_form_interaction = \"\"\"\ndocument.querySelector('#your-search').value = 'TypeScript commits';\ndocument.querySelector('form').submit();\n\"\"\"\n\nconfig = CrawlerRunConfig(…", "filename": ""}], "chunk_position": 50, "heading_path": "3. Handling Dynamic Content > 3. Handling Dynamic Content", "breadcrumbs": "Page Interaction - Crawl4AI Documentation (v0.9.x) > 3. Handling Dynamic Content > 3. Handling Dynamic Content"}, {"id": "55e68d54ca035639", "url": "https://docs.crawl4ai.com/core/page-interaction/", "page_title": "Page Interaction - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains how to interact with dynamic webpages using Crawl4AI, covering JavaScript execution, wait conditions, multi-step flows, Shadow DOM flattening, and virtual scrolling.", "heading": "4. Timing Control", "content": "Page: Page Interaction - Crawl4AI Documentation (v0.9.x)\nSection: 4. Timing Control\n\n1. **`page_timeout`** (ms): Overall page load or script execution time limit.\n2. **`delay_before_return_html`** (seconds): Wait an extra moment before capturing the final HTML.\n3. **`mean_delay`** &…", "code_blocks": [{"language": "python", "code": "config = CrawlerRunConfig(\n    page_timeout=60000,  # 60s limit\n    delay_before_return_html=2.5\n)", "filename": ""}], "chunk_position": 50, "heading_path": "4. Timing Control > 4. Timing Control", "breadcrumbs": "Page Interaction - Crawl4AI Documentation (v0.9.x) > 4. Timing Control > 4. Timing Control"}, {"id": "424e197600878c35", "url": "https://docs.crawl4ai.com/core/page-interaction/", "page_title": "Page Interaction - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains how to interact with dynamic webpages using Crawl4AI, covering JavaScript execution, wait conditions, multi-step flows, Shadow DOM flattening, and virtual scrolling.", "heading": "5. Multi-Step Interaction Example", "content": "Page: Page Interaction - Crawl4AI Documentation (v0.9.x)\nSection: 5. Multi-Step Interaction Example\n\nBelow is a simplified script that does multiple “Load More” clicks on GitHub’s TypeScript commits page. It **re-uses** the same session to accumulate new commits each time. The code includes the…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode\n\nasync def multi_page_commits():\n    browser_cfg = BrowserConfig(\n        headless=False,  # Visible…", "filename": ""}], "chunk_position": 50, "heading_path": "5. Multi-Step Interaction Example > 5. Multi-Step Interaction Example", "breadcrumbs": "Page Interaction - Crawl4AI Documentation (v0.9.x) > 5. Multi-Step Interaction Example > 5. Multi-Step Interaction Example"}, {"id": "00ed56a767cf1ce6", "url": "https://docs.crawl4ai.com/core/page-interaction/", "page_title": "Page Interaction - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains how to interact with dynamic webpages using Crawl4AI, covering JavaScript execution, wait conditions, multi-step flows, Shadow DOM flattening, and virtual scrolling.", "heading": "6. Combine Interaction with Extraction", "content": "Page: Page Interaction - Crawl4AI Documentation (v0.9.x)\nSection: 6. Combine Interaction with Extraction\n\nOnce dynamic content is loaded, you can attach an **`extraction_strategy`** (like `JsonCssExtractionStrategy` or `LLMExtractionStrategy`). For example:\n\nWhen done, check `result.extracted_content`…", "code_blocks": [{"language": "python", "code": "from crawl4ai import JsonCssExtractionStrategy\n\nschema = {\n    \"name\": \"Commits\",\n    \"baseSelector\": \"li.Box-sc-g0xbh4-0\",\n    \"fields\": [\n        {\"name\": \"title\", \"selector\": \"h4.markdown-title\",…", "filename": ""}], "chunk_position": 50, "heading_path": "6. Combine Interaction with Extraction > 6. Combine Interaction with Extraction", "breadcrumbs": "Page Interaction - Crawl4AI Documentation (v0.9.x) > 6. Combine Interaction with Extraction > 6. Combine Interaction with Extraction"}, {"id": "e57658260d72ca03", "url": "https://docs.crawl4ai.com/core/page-interaction/", "page_title": "Page Interaction - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains how to interact with dynamic webpages using Crawl4AI, covering JavaScript execution, wait conditions, multi-step flows, Shadow DOM flattening, and virtual scrolling.", "heading": "7. Shadow DOM Flattening", "content": "Page: Page Interaction - Crawl4AI Documentation (v0.9.x)\nSection: 7. Shadow DOM Flattening\n\nSites built with **Web Components** (Stencil, Lit, Shoelace, etc.) render content inside Shadow DOM — an encapsulated sub-tree that is invisible to normal page serialization. Set…", "code_blocks": [{"language": "python", "code": "config = CrawlerRunConfig(\n    flatten_shadow_dom=True,\n    wait_until=\"load\",\n    delay_before_return_html=3.0,  # give components time to hydrate\n)", "filename": ""}], "chunk_position": 50, "heading_path": "7. Shadow DOM Flattening > 7. Shadow DOM Flattening", "breadcrumbs": "Page Interaction - Crawl4AI Documentation (v0.9.x) > 7. Shadow DOM Flattening > 7. Shadow DOM Flattening"}, {"id": "279de492093a4840", "url": "https://docs.crawl4ai.com/core/page-interaction/", "page_title": "Page Interaction - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains how to interact with dynamic webpages using Crawl4AI, covering JavaScript execution, wait conditions, multi-step flows, Shadow DOM flattening, and virtual scrolling.", "heading": "8. Relevant `CrawlerRunConfig` Parameters", "content": "Page: Page Interaction - Crawl4AI Documentation (v0.9.x)\nSection: 8. Relevant `CrawlerRunConfig` Parameters\n\nBelow are the key interaction-related parameters in `CrawlerRunConfig`. For a full list, see [Configuration Parameters](../../api/parameters/).\n\n- **`js_code`**: JavaScript to run after `wait_for` +…", "code_blocks": [], "chunk_position": 50, "heading_path": "8. Relevant `CrawlerRunConfig` Parameters > 8. Relevant `CrawlerRunConfig` Parameters", "breadcrumbs": "Page Interaction - Crawl4AI Documentation (v0.9.x) > 8. Relevant `CrawlerRunConfig` Parameters > 8. Relevant `CrawlerRunConfig` Parameters"}, {"id": "c3977b7660922ce4", "url": "https://docs.crawl4ai.com/core/page-interaction/", "page_title": "Page Interaction - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains how to interact with dynamic webpages using Crawl4AI, covering JavaScript execution, wait conditions, multi-step flows, Shadow DOM flattening, and virtual scrolling.", "heading": "9. Conclusion", "content": "Page: Page Interaction - Crawl4AI Documentation (v0.9.x)\nSection: 9. Conclusion\n\nCrawl4AI's **page interaction** features let you:\n\n1. **Execute JavaScript** for scrolling, clicks, or form filling.\n2. **Wait** for CSS or custom JS conditions before capturing data.\n3. **Handle**…", "code_blocks": [], "chunk_position": 50, "heading_path": "9. Conclusion > 9. Conclusion", "breadcrumbs": "Page Interaction - Crawl4AI Documentation (v0.9.x) > 9. Conclusion > 9. Conclusion"}, {"id": "0b1ee59b45db6343", "url": "https://docs.crawl4ai.com/core/page-interaction/", "page_title": "Page Interaction - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page explains how to interact with dynamic webpages using Crawl4AI, covering JavaScript execution, wait conditions, multi-step flows, Shadow DOM flattening, and virtual scrolling.", "heading": "10. Virtual Scrolling", "content": "Page: Page Interaction - Crawl4AI Documentation (v0.9.x)\nSection: 10. Virtual Scrolling\n\nFor sites that use **virtual scrolling** (where content is replaced rather than appended as you scroll, like Twitter or Instagram), Crawl4AI provides a dedicated `VirtualScrollConfig`:\n\n###### Virtual…", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncWebCrawler, CrawlerRunConfig, VirtualScrollConfig\n\nasync def crawl_twitter_timeline():\n    # Configure virtual scroll for Twitter-like feeds\n    virtual_config =…", "filename": ""}], "chunk_position": 50, "heading_path": "10. Virtual Scrolling > 10. Virtual Scrolling", "breadcrumbs": "Page Interaction - Crawl4AI Documentation (v0.9.x) > 10. Virtual Scrolling > 10. Virtual Scrolling"}, {"id": "72530d3f20da23dd", "url": "https://docs.crawl4ai.com/core/quickstart/", "page_title": "Quick Start - Crawl4AI Documentation (v0.9.x)", "page_type": "reference", "page_summary": "Extraction fallback content.", "heading": "Quick Start - Crawl4AI Documentation (v0.9.x)", "content": "Page: Quick Start - Crawl4AI Documentation (v0.9.x)\nSection: Quick Start - Crawl4AI Documentation (v0.9.x)\n\n\n\n\n##### Getting Started with Crawl4AI\n\n\n\n\nWelcome to  **Crawl4AI** , an open-source LLM-friendly Web Crawler & Scraper. In this tutorial, you’ll:\n\n\n\n\n\n\n- Run your  **first crawl**  using minimal configuration.\n\n- Generate  **Markdown**  output (and learn how it’s influenced by content filters).\n\n- Experiment with a simple  **CSS-based extraction**  strategy.\n\n- See a glimpse of  **LLM-based extraction**  (including open-source and closed-source model options).\n\n- Crawl a  **dynamic**  page that loads content via JavaScript.\n\n\n\n\n\n\n##### 1. Introduction\n\n\n\n\nCrawl4AI provides:\n\n\n\n\n\n\n- An asynchronous crawler,  **`AsyncWebCrawler`** .\n\n- Configurable browser and run settings via  **`BrowserConfig`**  and  **`CrawlerRunConfig`** .\n\n- Automatic HTML-to-Markdown conversion via  **`DefaultMarkdownGenerator`**  (supports optional filters).\n\n- Multiple extraction strategies (LLM-based or “traditional” CSS/XPath-based).\n\n\n\n\n\nBy the end of this guide, you’ll have performed a basic crawl, generated Markdown, tried out two extraction strategies, and crawled a dynamic page that uses “Load More” buttons or JavaScript updates.\n\n\n\n\n\n##### 2. Your First Crawl\n\n\n\n\nHere’s a minimal Python script that creates an  **`AsyncWebCrawler`** , fetches a webpage, and prints the first 300 characters of its Markdown output:\n\n\n\n\n\n **What’s happening?** \n-  **`AsyncWebCrawler`**  launches a headless browser (Chromium by default).\n- It fetches `https://example.com`.\n- Crawl4AI automatically converts the HTML into Markdown.\n\n\n\n\nYou now have a simple, working crawl!\n\n\n\n\n\n##### 3. Basic Configuration (Light Introduction)\n\n\n\n\nCrawl4AI’s crawler can be heavily customized using two main classes:\n\n\n\n\n1.  **`BrowserConfig`** : Controls browser behavior (headless or full UI, user agent, JavaScript toggles, etc.).\n\n2.  **`CrawlerRunConfig`** : Controls how each crawl runs (caching, extraction, timeouts, hooking, etc.).\n\n\n\n\nBelow is an example with minimal usage:\n\n\n\n\n\n> IMPORTANT: By default cache mode is set to `CacheMode.BYPASS` to have fresh content. Set `CacheMode.ENABLED` to enable caching.\n\n\n\n\nWe’ll explore more advanced config in later tutorials (like enabling proxies, PDF output, multi-tab sessions, etc.). For now, just note how you pass these objects to manage crawling.\n\n\n\n\n\n##### 4. Generating Markdown Output\n\n\n\n\nBy default, Crawl4AI automatically generates Markdown from each crawled page. However, the exact output depends on whether you specify a  **markdown generator**  or  **content filter** .\n\n\n\n\n\n\n- **`result.markdown`** :\n\n  The direct HTML-to-Markdown conversion.\n\n- **`result.markdown.fit_markdown`** :\n\n  The same content after applying any configured  **content filter**  (e.g., `PruningContentFilter`).\n\n\n\n\n\n###### Example: Using a Filter with `DefaultMarkdownGenerator`\n\n\n\n\n\n **Note** : If you do  **not**  specify a content filter or markdown generator, you’ll typically see only the raw Markdown. `PruningContentFilter` may adds around `50ms` in processing time. We’ll dive deeper into these strategies in a dedicated  **Markdown Generation**  tutorial.\n\n\n\n\n\n##### 5. Simple Data Extraction (CSS-based)\n\n\n\n\nCrawl4AI can also extract structured data (JSON) using CSS or XPath selectors. Below is a minimal CSS-based example:\n\n\n\n\n> **New!**  Crawl4AI now provides a powerful utility to automatically generate extraction schemas using LLM. This is a one-time cost that gives you a reusable schema for fast, LLM-free extractions:\n\n\n\n\n\nFor a complete guide on schema generation and advanced usage, see [No-LLM Extraction Strategies](../../extraction/no-llm-strategies/).\n\n\n\n\nHere's a basic extraction example:\n\n\n\n\n\n **Why is this helpful?** \n- Great for repetitive page structures (e.g., item listings, articles).\n- No AI usage or costs.\n- The crawler returns a JSON string you can parse or store.\n\n\n\n\n> Tips: You can pass raw HTML to the crawler instead of a URL. To do so, prefix the HTML with `raw://`.\n\n\n\n\n\n##### 6. Simple Data Extraction (LLM-based)\n\n\n\n\nFor more complex or irregular pages, a language model can parse text intelligently into a structure you define. Crawl4AI supports  **open-source**  or  **closed-source**  providers:\n\n\n\n\n\n\n- **Open-Source Models**  (e.g., `ollama/llama3.3`, `no_token`)\n\n- **OpenAI Models**  (e.g., `openai/gpt-4`, requires `api_token`)\n\n- Or any provider supported by the underlying library\n\n\n\n\n\nBelow is an example using  **open-source**  style (no token) and closed-source:\n\n\n\n\n\n **What’s happening?** \n- We define a Pydantic schema (`PricingInfo`) describing the fields we want.\n- The LLM extraction strategy uses that schema and your instructions to transform raw text into structured JSON.\n- Depending on the  **provider**  and  **api_token** , you can use local models or a remote API.\n\n\n\n\n\n##### 7. Adaptive Crawling (New!)\n\n\n\n\nCrawl4AI now includes intelligent adaptive crawling that automatically determines when sufficient information has been gathered. Here's a quick example:\n\n\n\n\n\n **What's special about adaptive crawling?** \n-  **Automatic stopping** : Stops when sufficient information is gathered\n-  **Intelligent link selection** : Follows only relevant links\n-  **Confidence scoring** : Know how complete your information is\n\n\n\n\n[Learn more about Adaptive Crawling →](../adaptive-crawling/)\n\n\n\n\n\n##### 8. Multi-URL Concurrency (Preview)\n\n\n\n\nIf you need to crawl multiple URLs in  **parallel** , you can use `arun_many()`. By default, Crawl4AI employs a  **MemoryAdaptiveDispatcher** , automatically adjusting concurrency based on system resources. Here’s a quick glimpse:\n\n\n\n\n\nThe example above shows two ways to handle multiple URLs:\n1.  **Streaming mode**  (`stream=True`): Process results as they become available using `async for`\n2.  **Batch mode**  (`stream=False`): Wait for all results to complete\n\n\n\n\nFor more advanced concurrency (e.g., a  **semaphore-based**  approach,  **adaptive memory usage throttling** , or customized rate limiting), see [Advanced Multi-URL Crawling](../../advanced/multi-url-crawling/).\n\n\n\n\n\n##### 8. Dynamic Content Example\n\n\n\n\nSome sites require multiple “page clicks” or dynamic JavaScript updates. Below is an example showing how to  **click**  a “Next Page” button and wait for new commits to load on GitHub, using  **`BrowserConfig`**  and  **`CrawlerRunConfig`** :\n\n\n\n\n\n **Key Points** :\n\n\n\n\n\n\n- **`BrowserConfig(headless=False)`** : We want to watch it click “Next Page.”\n\n- **`CrawlerRunConfig(...)`** : We specify the extraction strategy, pass `session_id` to reuse the same page.\n\n- **`js_code`**  and  **`wait_for`**  are used for subsequent pages (`page > 0`) to click the “Next” button and wait for new commits to load.\n\n- **`js_only=True`**  indicates we’re not re-navigating but continuing the existing session.\n\n- Finally, we call `kill_session()` to clean up the page and browser session.\n\n\n\n\n\n\n##### 9. Next Steps\n\n\n\n\nCongratulations! You have:\n\n\n\n\n\n\n- Performed a basic crawl and printed Markdown.\n\n- Used  **content filters**  with a markdown generator.\n\n- Extracted JSON via  **CSS**  or  **LLM**  strategies.\n\n- Handled  **dynamic**  pages with JavaScript triggers.\n\n\n\n\n\nIf you’re ready for more, check out:\n\n\n\n\n\n\n- **Installation** : A deeper dive into advanced installs, Docker usage (experimental), or optional dependencies.\n\n- **Hooks & Auth** : Learn how to run custom JavaScript or handle logins with cookies, local storage, etc.\n\n- **Deployment** : Explore ephemeral testing in Docker or plan for the upcoming stable Docker release.\n\n- **Browser Management** : Delve into user simulation, stealth modes, and concurrency best practices.\n\n\n\n\n\nCrawl4AI is a powerful, flexible tool. Enjoy building out your scrapers, data pipelines, or AI-driven extraction flows. Happy crawling!\n\n\n\nPage Copy\nPage Copy\n\n\n\n\n- [Copy as Markdown\nCopy page for LLMs](#)\n\n- [View as Markdown\nOpen raw source](#)\n\n\n- [Open in ChatGPT\nAsk questions about this page](#)\n\n\n\nESC to close\n", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler\n\nasync def main():\n    async with AsyncWebCrawler() as crawler:\n        result = await crawler.arun(\"https://example.com\")\n        print(result.markdown[:300])  # Print first 300 chars\n\nif __name__ == \"__main__\":\n    asyncio.run(main())", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode\n\nasync def main():\n    browser_conf = BrowserConfig(headless=True)  # or False to see the browser\n    run_conf = CrawlerRunConfig(\n        cache_mode=CacheMode.BYPASS\n    )\n\n    async with AsyncWebCrawler(config=browser_conf) as crawler:\n        result = await crawler.arun(\n            url=\"https://example.com\",\n            config=run_conf\n        )\n        print(result.markdown)\n\nif __name__ == \"__main__\":\n    asyncio.run(main())", "filename": ""}, {"language": "python", "code": "from crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\nfrom crawl4ai.content_filter_strategy import PruningContentFilter\nfrom crawl4ai.markdown_generation_strategy import DefaultMarkdownGenerator\n\nmd_generator = DefaultMarkdownGenerator(\n    content_filter=PruningContentFilter(threshold=0.4, threshold_type=\"fixed\")\n)\n\nconfig = CrawlerRunConfig(\n    cache_mode=CacheMode.BYPASS,\n    markdown_generator=md_generator\n)\n\nasync with AsyncWebCrawler() as crawler:\n    result = await crawler.arun(\"https://news.ycombinator.com\", config=config)\n    print(\"Raw Markdown length:\", len(result.markdown.raw_markdown))\n    print(\"Fit Markdown length:\", len(result.markdown.fit_markdown))", "filename": ""}, {"language": "python", "code": "from crawl4ai import JsonCssExtractionStrategy\nfrom crawl4ai import LLMConfig\n\n# Generate a schema (one-time cost)\nhtml = \"<div class='product'><h2>Gaming Laptop</h2><span class='price'>$999.99</span></div>\"\n\n# Using OpenAI (requires API token)\nschema = JsonCssExtractionStrategy.generate_schema(\n    html,\n    llm_config = LLMConfig(provider=\"openai/gpt-4o\",api_token=\"your-openai-token\")  # Required for OpenAI\n)\n\n# Or using Ollama (open source, no token needed)\nschema = JsonCssExtractionStrategy.generate_schema(\n    html,\n    llm_config = LLMConfig(provider=\"ollama/llama3.3\", api_token=None)  # Not needed for Ollama\n)\n\n# Use the schema for fast, repeated extractions\nstrategy = JsonCssExtractionStrategy(schema)", "filename": ""}, {"language": "python", "code": "import asyncio\nimport json\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\nfrom crawl4ai import JsonCssExtractionStrategy\n\nasync def main():\n    schema = {\n        \"name\": \"Example Items\",\n        \"baseSelector\": \"div.item\",\n        \"fields\": [\n            {\"name\": \"title\", \"selector\": \"h2\", \"type\": \"text\"},\n            {\"name\": \"link\", \"selector\": \"a\", \"type\": \"attribute\", \"attribute\": \"href\"}\n        ]\n    }\n\n    raw_html = \"<div class='item'><h2>Item 1</h2><a href='https://example.com/item1'>Link 1</a></div>\"\n\n    async with AsyncWebCrawler() as crawler:\n        result = await crawler.arun(\n            url=\"raw://\" + raw_html,\n            config=CrawlerRunConfig(\n                cache_mode=CacheMode.BYPASS,\n                extraction_strategy=JsonCssExtractionStrategy(schema)\n            )\n        )\n        # The JSON output is stored in 'extracted_content'\n        data = json.loads(result.extracted_content)\n        print(data)\n\nif __name__ == \"__main__\":\n    asyncio.run(main())", "filename": ""}, {"language": "python", "code": "import os\nimport json\nimport asyncio\nfrom pydantic import BaseModel, Field\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, LLMConfig\nfrom crawl4ai import LLMExtractionStrategy\n\nclass OpenAIModelFee(BaseModel):\n    model_name: str = Field(..., description=\"Name of the OpenAI model.\")\n    input_fee: str = Field(..., description=\"Fee for input token for the OpenAI model.\")\n    output_fee: str = Field(\n        ..., description=\"Fee for output token for the OpenAI model.\"\n    )\n\nasync def extract_structured_data_using_llm(\n    provider: str, api_token: str = None, extra_headers: Dict[str, str] = None\n):\n    print(f\"\\n--- Extracting Structured Data with {provider} ---\")\n\n    if api_token is None and provider != \"ollama\":\n        print(f\"API token is required for {provider}. Skipping this example.\")\n        return\n\n    browser_config = BrowserConfig(headless=True)\n\n    extra_args = {\"temperature\": 0, \"top_p\": 0.9, \"max_tokens\": 2000}\n    if extra_headers:\n        extra_args[\"extra_headers\"] = extra_headers\n\n    crawler_config = CrawlerRunConfig(\n        cache_mode=CacheMode.BYPASS,\n        word_count_threshold=1,\n        page_timeout=80000,\n        extraction_strategy=LLMExtractionStrategy(\n            llm_config = LLMConfig(provider=provider,api_token=api_token),\n            schema=OpenAIModelFee.model_json_schema(),\n            extraction_type=\"schema\",\n            instruction=\"\"\"From the crawled content, extract all mentioned model names along with their fees for input and output tokens. \n            Do not miss any models in the entire content.\"\"\",\n            extra_args=extra_args,\n        ),\n    )\n\n    async with AsyncWebCrawler(config=browser_config) as crawler:\n        result = await crawler.arun(\n            url=\"https://openai.com/api/pricing/\", config=crawler_config\n        )\n        print(result.extracted_content)\n\nif __name__ == \"__main__\":\n\n    asyncio.run(\n        extract_structured_data_using_llm(\n            provider=\"openai/gpt-4o\", api_token=os.getenv(\"OPENAI_API_KEY\")\n        )\n    )", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, AdaptiveCrawler\n\nasync def adaptive_example():\n    async with AsyncWebCrawler() as crawler:\n        adaptive = AdaptiveCrawler(crawler)\n\n        # Start adaptive crawling\n        result = await adaptive.digest(\n            start_url=\"https://docs.python.org/3/\",\n            query=\"async context managers\"\n        )\n\n        # View results\n        adaptive.print_stats()\n        print(f\"Crawled {len(result.crawled_urls)} pages\")\n        print(f\"Achieved {adaptive.confidence:.0%} confidence\")\n\nif __name__ == \"__main__\":\n    asyncio.run(adaptive_example())", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\n\nasync def quick_parallel_example():\n    urls = [\n        \"https://example.com/page1\",\n        \"https://example.com/page2\",\n        \"https://example.com/page3\"\n    ]\n\n    run_conf = CrawlerRunConfig(\n        cache_mode=CacheMode.BYPASS,\n        stream=True  # Enable streaming mode\n    )\n\n    async with AsyncWebCrawler() as crawler:\n        # Stream results as they complete\n        async for result in await crawler.arun_many(urls, config=run_conf):\n            if result.success:\n                print(f\"[OK] {result.url}, length: {len(result.markdown.raw_markdown)}\")\n            else:\n                print(f\"[ERROR] {result.url} => {result.error_message}\")\n\n        # Or get all results at once (default behavior)\n        run_conf = run_conf.clone(stream=False)\n        results = await crawler.arun_many(urls, config=run_conf)\n        for res in results:\n            if res.success:\n                print(f\"[OK] {res.url}, length: {len(res.markdown.raw_markdown)}\")\n            else:\n                print(f\"[ERROR] {res.url} => {res.error_message}\")\n\nif __name__ == \"__main__\":\n    asyncio.run(quick_parallel_example())", "filename": ""}, {"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode\nfrom crawl4ai import JsonCssExtractionStrategy\n\nasync def extract_structured_data_using_css_extractor():\n    print(\"\\n--- Using JsonCssExtractionStrategy for Fast Structured Output ---\")\n    schema = {\n        \"name\": \"KidoCode Courses\",\n        \"baseSelector\": \"section.charge-methodology .w-tab-content > div\",\n        \"fields\": [\n            {\n                \"name\": \"section_title\",\n                \"selector\": \"h3.heading-50\",\n                \"type\": \"text\",\n            },\n            {\n                \"name\": \"section_description\",\n                \"selector\": \".charge-content\",\n                \"type\": \"text\",\n            },\n            {\n                \"name\": \"course_name\",\n                \"selector\": \".text-block-93\",\n                \"type\": \"text\",\n            },\n            {\n                \"name\": \"course_description\",\n                \"selector\": \".course-content-text\",\n                \"type\": \"text\",\n            },\n            {\n                \"name\": \"course_icon\",\n                \"selector\": \".image-92\",\n                \"type\": \"attribute\",\n                \"attribute\": \"src\",\n            },\n        ],\n    }\n\n    browser_config = BrowserConfig(headless=True, java_script_enabled=True)\n\n    js_click_tabs = \"\"\"\n    (async () => {\n        const tabs = document.querySelectorAll(\"section.charge-methodology .tabs-menu-3 > div\");\n        for(let tab of tabs) {\n            tab.scrollIntoView();\n            tab.click();\n            await new Promise(r => setTimeout(r, 500));\n        }\n    })();\n    \"\"\"\n\n    crawler_config = CrawlerRunConfig(\n        cache_mode=CacheMode.BYPASS,\n        extraction_strategy=JsonCssExtractionStrategy(schema),\n        js_code=[js_click_tabs],\n    )\n\n    async with AsyncWebCrawler(config=browser_config) as crawler:\n        result = await crawler.arun(\n            url=\"https://www.kidocode.com/degrees/technology\", config=crawler_config\n        )\n\n        companies = json.loads(result.extracted_content)\n        print(f\"Successfully extracted {len(companies)} companies\")\n        print(json.dumps(companies[0], indent=2))\n\nasync def main():\n    await extract_structured_data_using_css_extractor()\n\nif __name__ == \"__main__\":\n    asyncio.run(main())", "filename": ""}], "chunk_position": 51, "heading_path": "Quick Start - Crawl4AI Documentation (v0.9.x) > Quick Start - Crawl4AI Documentation (v0.9.x)", "breadcrumbs": "Quick Start - Crawl4AI Documentation (v0.9.x) > Quick Start - Crawl4AI Documentation (v0.9.x) > Quick Start - Crawl4AI Documentation (v0.9.x)"}, {"id": "88729416cff4149f", "url": "https://docs.crawl4ai.com/core/simple-crawling/", "page_title": "Simple Crawling", "page_type": "guide", "page_summary": "This guide covers the basics of web crawling with Crawl4AI, including setting up a crawler, making requests, understanding responses, and handling errors.", "heading": "Basic Usage", "content": "Page: Simple Crawling\nSection: Basic Usage\n\nSet up a simple crawl using `BrowserConfig` and `CrawlerRunConfig`:", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler\nfrom crawl4ai.async_configs import BrowserConfig, CrawlerRunConfig\n\nasync def main():\n    browser_config = BrowserConfig()  # Default browser…", "filename": ""}], "chunk_position": 52, "heading_path": "Basic Usage > Basic Usage", "breadcrumbs": "Simple Crawling > Basic Usage > Basic Usage"}, {"id": "af391d0924c319df", "url": "https://docs.crawl4ai.com/core/simple-crawling/", "page_title": "Simple Crawling", "page_type": "guide", "page_summary": "This guide covers the basics of web crawling with Crawl4AI, including setting up a crawler, making requests, understanding responses, and handling errors.", "heading": "Understanding the Response", "content": "Page: Simple Crawling\nSection: Understanding the Response\n\nThe `arun()` method returns a `CrawlResult` object with several useful properties. Here's a quick overview (see [CrawlResult](../../api/crawl-result/) for complete details):", "code_blocks": [{"language": "python", "code": "config = CrawlerRunConfig(\n    markdown_generator=DefaultMarkdownGenerator(\n        content_filter=PruningContentFilter(threshold=0.6),\n        options={\"ignore_links\": True}\n    )\n)\n\nresult = await…", "filename": ""}], "chunk_position": 52, "heading_path": "Understanding the Response > Understanding the Response", "breadcrumbs": "Simple Crawling > Understanding the Response > Understanding the Response"}, {"id": "bbcf2f34a8618ff4", "url": "https://docs.crawl4ai.com/core/simple-crawling/", "page_title": "Simple Crawling", "page_type": "guide", "page_summary": "This guide covers the basics of web crawling with Crawl4AI, including setting up a crawler, making requests, understanding responses, and handling errors.", "heading": "Complete Example", "content": "Page: Simple Crawling\nSection: Complete Example\n\nHere's a more comprehensive example demonstrating common usage patterns:", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler\nfrom crawl4ai.async_configs import BrowserConfig, CrawlerRunConfig, CacheMode\n\nasync def main():\n    browser_config = BrowserConfig(verbose=True)…", "filename": ""}], "chunk_position": 52, "heading_path": "Complete Example > Complete Example", "breadcrumbs": "Simple Crawling > Complete Example > Complete Example"}, {"id": "20bf17caac1c6bc4", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Overview", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Overview\n\n**New in v0.7.3+** : Table extraction now follows the **Strategy Design Pattern** , providing unprecedented flexibility and power for handling different table structures. Don't worry - **your…", "code_blocks": [], "chunk_position": 53, "heading_path": "Overview > Overview", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Overview > Overview"}, {"id": "b54665b525c3e9ed", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "What's Changed?", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: What's Changed?\n\n- **Architecture** : Table extraction now uses pluggable strategies\n- **Backward Compatible** : Your existing code with `table_score_threshold` continues to work\n- **More Power** : Choose from…", "code_blocks": [], "chunk_position": 53, "heading_path": "What's Changed? > What's Changed?", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > What's Changed? > What's Changed?"}, {"id": "99a0a0457d00ef66", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Key Points", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Key Points\n\n✅ **Old code still works** - No breaking changes\n✅ **Same default behavior** - Uses the proven extraction algorithm\n✅ **New capabilities** - Add LLM extraction or custom strategies when needed\n✅…", "code_blocks": [], "chunk_position": 53, "heading_path": "Key Points > Key Points", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Key Points > Key Points"}, {"id": "850a7d2da0760666", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "The Simplest Way (Works Like Before)", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: The Simplest Way (Works Like Before)\n\nIf you're already using Crawl4AI, nothing changes:", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\n\nasync def extract_tables():\n    async with AsyncWebCrawler() as crawler:\n        # This works exactly like before - uses…", "filename": ""}], "chunk_position": 53, "heading_path": "The Simplest Way (Works Like Before) > The Simplest Way (Works Like Before)", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > The Simplest Way (Works Like Before) > The Simplest Way (Works Like Before)"}, {"id": "ff6413ac3ffd6166", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Using the Old Configuration (Still Supported)", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Using the Old Configuration (Still Supported)\n\nYour existing code with `table_score_threshold` continues to work:", "code_blocks": [{"language": "python", "code": "# This old approach STILL WORKS - we maintain backward compatibility\nconfig = CrawlerRunConfig(\n    table_score_threshold=7  # Internally creates…", "filename": ""}], "chunk_position": 53, "heading_path": "Using the Old Configuration (Still Supported) > Using the Old Configuration (Still Supported)", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Using the Old Configuration (Still Supported) > Using the Old Configuration (Still Supported)"}, {"id": "73df462da0ead460", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Understanding the Strategy Pattern", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Understanding the Strategy Pattern\n\nThe strategy pattern allows you to choose different table extraction algorithms at runtime. Think of it as having different tools in a toolbox - you pick the right one for the job:\n\n- **No explicit…", "code_blocks": [], "chunk_position": 53, "heading_path": "Understanding the Strategy Pattern > Understanding the Strategy Pattern", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Understanding the Strategy Pattern > Understanding the Strategy Pattern"}, {"id": "7f3a564d74958426", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Available Strategies", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Available Strategies\n\n| Strategy | Description | Use Case | Cost | When to Use |\n| --- | --- | --- | --- | --- |\n| `DefaultTableExtraction` | **RECOMMENDED** : Same algorithm as before v0.7.3 | General purpose (default) |…", "code_blocks": [], "chunk_position": 53, "heading_path": "Available Strategies > Available Strategies", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Available Strategies > Available Strategies"}, {"id": "4a2b439249556a30", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "DefaultTableExtraction", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: DefaultTableExtraction\n\nThe default strategy uses a sophisticated scoring system to identify data tables:", "code_blocks": [{"language": "python", "code": "from crawl4ai import DefaultTableExtraction, CrawlerRunConfig\n\n# Customize the default extraction\ntable_strategy = DefaultTableExtraction(\n    table_score_threshold=7,  # Scoring threshold (default:…", "filename": ""}], "chunk_position": 53, "heading_path": "DefaultTableExtraction > DefaultTableExtraction", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > DefaultTableExtraction > DefaultTableExtraction"}, {"id": "377b39fb667715ba", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Scoring System", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Scoring System\n\nThe scoring system evaluates multiple factors:\n\n| Factor | Score Impact | Description |\n| --- | --- | --- |\n| Has `<thead>` | +2 | Semantic table structure |\n| Has `<tbody>` | +1 | Organized table…", "code_blocks": [], "chunk_position": 53, "heading_path": "Scoring System > Scoring System", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Scoring System > Scoring System"}, {"id": "b8e2860b68b977da", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "LLMTableExtraction (Use Sparingly!)", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: LLMTableExtraction (Use Sparingly!)\n\n**⚠️ WARNING** : Only use this when `DefaultTableExtraction` fails with complex tables!\n\nLLMTableExtraction uses AI to understand complex table structures that traditional parsers struggle with. It…", "code_blocks": [{"language": "python", "code": "from crawl4ai import LLMTableExtraction, LLMConfig, CrawlerRunConfig\n\n# Configure LLM (costs money per call!)\nllm_config = LLMConfig(\n    provider=\"groq/llama-3.3-70b-versatile\",  # Fast provider for…", "filename": ""}], "chunk_position": 53, "heading_path": "LLMTableExtraction (Use Sparingly!) > LLMTableExtraction (Use Sparingly!)", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > LLMTableExtraction (Use Sparingly!) > LLMTableExtraction (Use Sparingly!)"}, {"id": "33e5721edf75e122", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "When to Use LLMTableExtraction", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: When to Use LLMTableExtraction\n\n✅ **Use ONLY when** :\n- Tables have complex merged cells (rowspan/colspan) that break DefaultTableExtraction\n- Nested tables that need semantic understanding\n- Tables with irregular structures\n-…", "code_blocks": [], "chunk_position": 53, "heading_path": "When to Use LLMTableExtraction > When to Use LLMTableExtraction", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > When to Use LLMTableExtraction > When to Use LLMTableExtraction"}, {"id": "026b21534c2671a6", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "How Smart Chunking Works", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: How Smart Chunking Works\n\nLLMTableExtraction automatically handles large tables through intelligent chunking:\n\n- **Automatic Detection** : Tables exceeding the token threshold are automatically split\n- **Smart Splitting** :…", "code_blocks": [], "chunk_position": 53, "heading_path": "How Smart Chunking Works > How Smart Chunking Works", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > How Smart Chunking Works > How Smart Chunking Works"}, {"id": "fbe23ad2b4422f8c", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Performance Optimization for LLMTableExtraction", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Performance Optimization for LLMTableExtraction\n\n**Provider Recommendations by Table Size** :\n\n| Table Size | Recommended Providers | Why |\n| --- | --- | --- |\n| Small (<50 rows) | Any provider | Fast enough |\n| Medium (50-200 rows) | Groq,…", "code_blocks": [], "chunk_position": 53, "heading_path": "Performance Optimization for LLMTableExtraction > Performance Optimization for LLMTableExtraction", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Performance Optimization for LLMTableExtraction > Performance Optimization for LLMTableExtraction"}, {"id": "2b7852c5fef82b2c", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "NoTableExtraction", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: NoTableExtraction\n\nDisable table extraction for better performance when tables aren't needed:", "code_blocks": [{"language": "python", "code": "from crawl4ai import NoTableExtraction, CrawlerRunConfig\n\nconfig = CrawlerRunConfig(\n    table_extraction=NoTableExtraction()\n)\n\n# Tables won't be extracted, improving performance\nresult = await…", "filename": ""}], "chunk_position": 53, "heading_path": "NoTableExtraction > NoTableExtraction", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > NoTableExtraction > NoTableExtraction"}, {"id": "e62beb22b2c20458", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Basic Configuration", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Basic Configuration\n\n```python\nconfig = CrawlerRunConfig(\n    # Table extraction settings\n    table_score_threshold=7,      # Default threshold (backward compatible)\n    table_extraction=strategy,     # Optional: custom strategy…\n```", "code_blocks": [{"language": "python", "code": "config = CrawlerRunConfig(\n    # Table extraction settings\n    table_score_threshold=7,      # Default threshold (backward compatible)\n    table_extraction=strategy,     # Optional: custom strategy…", "filename": ""}], "chunk_position": 53, "heading_path": "Basic Configuration > Basic Configuration", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Basic Configuration > Basic Configuration"}, {"id": "1f6f70b71eaaa0f7", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Advanced Configuration", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Advanced Configuration\n\n```python\nfrom crawl4ai import DefaultTableExtraction, CrawlerRunConfig\n\n# Fine-tuned extraction\nstrategy = DefaultTableExtraction(\n    table_score_threshold=5,      # Lower = more permissive\n    min_rows=3,…\n```", "code_blocks": [{"language": "python", "code": "from crawl4ai import DefaultTableExtraction, CrawlerRunConfig\n\n# Fine-tuned extraction\nstrategy = DefaultTableExtraction(\n    table_score_threshold=5,      # Lower = more permissive\n    min_rows=3,…", "filename": ""}], "chunk_position": 53, "heading_path": "Advanced Configuration > Advanced Configuration", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Advanced Configuration > Advanced Configuration"}, {"id": "f08a6167ff96c777", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Convert to Pandas DataFrame", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Convert to Pandas DataFrame\n\n```python\nimport pandas as pd\n\nasync def tables_to_dataframes(url):\n    async with AsyncWebCrawler() as crawler:\n        result = await crawler.arun(url)\n\n        dataframes = []\n        for table_data in…\n```", "code_blocks": [{"language": "python", "code": "import pandas as pd\n\nasync def tables_to_dataframes(url):\n    async with AsyncWebCrawler() as crawler:\n        result = await crawler.arun(url)\n\n        dataframes = []\n        for table_data in…", "filename": ""}], "chunk_position": 53, "heading_path": "Convert to Pandas DataFrame > Convert to Pandas DataFrame", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Convert to Pandas DataFrame > Convert to Pandas DataFrame"}, {"id": "981d91d16c7e735f", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Filter Tables by Criteria", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Filter Tables by Criteria\n\n```python\nasync def extract_large_tables(url):\n    async with AsyncWebCrawler() as crawler:\n        # Configure minimum size requirements\n        strategy = DefaultTableExtraction(\n            min_rows=10,…\n```", "code_blocks": [{"language": "python", "code": "async def extract_large_tables(url):\n    async with AsyncWebCrawler() as crawler:\n        # Configure minimum size requirements\n        strategy = DefaultTableExtraction(\n            min_rows=10,…", "filename": ""}], "chunk_position": 53, "heading_path": "Filter Tables by Criteria > Filter Tables by Criteria", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Filter Tables by Criteria > Filter Tables by Criteria"}, {"id": "973dcfec9acdadbc", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Export Tables to Different Formats", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Export Tables to Different Formats\n\n```python\nimport json\nimport csv\n\nasync def export_tables(url):\n    async with AsyncWebCrawler() as crawler:\n        result = await crawler.arun(url)\n\n        for i, table in enumerate(result.tables):…\n```", "code_blocks": [{"language": "python", "code": "import json\nimport csv\n\nasync def export_tables(url):\n    async with AsyncWebCrawler() as crawler:\n        result = await crawler.arun(url)\n\n        for i, table in enumerate(result.tables):…", "filename": ""}], "chunk_position": 53, "heading_path": "Export Tables to Different Formats > Export Tables to Different Formats", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Export Tables to Different Formats > Export Tables to Different Formats"}, {"id": "4e13f3d0cf5223e2", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Creating Custom Strategies", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Creating Custom Strategies\n\nExtend `TableExtractionStrategy` to create custom extraction logic:", "code_blocks": [], "chunk_position": 53, "heading_path": "Creating Custom Strategies > Creating Custom Strategies", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Creating Custom Strategies > Creating Custom Strategies"}, {"id": "6ba7f14d11120a53", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Example: Financial Table Extractor", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Example: Financial Table Extractor\n\n```python\nfrom crawl4ai import TableExtractionStrategy\nfrom typing import List, Dict, Any\nimport re\n\nclass FinancialTableExtractor(TableExtractionStrategy):\n    \"\"\"Extract tables containing financial…\n```", "code_blocks": [{"language": "python", "code": "from crawl4ai import TableExtractionStrategy\nfrom typing import List, Dict, Any\nimport re\n\nclass FinancialTableExtractor(TableExtractionStrategy):\n    \"\"\"Extract tables containing financial…", "filename": ""}], "chunk_position": 53, "heading_path": "Example: Financial Table Extractor > Example: Financial Table Extractor", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Example: Financial Table Extractor > Example: Financial Table Extractor"}, {"id": "4ffab6b490435d2d", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Example: Specific Table Extractor", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Example: Specific Table Extractor\n\n```python\nclass SpecificTableExtractor(TableExtractionStrategy):\n    \"\"\"Extract only tables matching specific criteria.\"\"\"\n\n    def __init__(self, \n                 required_headers=None,…\n```", "code_blocks": [{"language": "python", "code": "class SpecificTableExtractor(TableExtractionStrategy):\n    \"\"\"Extract only tables matching specific criteria.\"\"\"\n\n    def __init__(self, \n                 required_headers=None,…", "filename": ""}], "chunk_position": 53, "heading_path": "Example: Specific Table Extractor > Example: Specific Table Extractor", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Example: Specific Table Extractor > Example: Specific Table Extractor"}, {"id": "cc2e57f32f30a950", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Combining with Other Strategies", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Combining with Other Strategies\n\nTable extraction works seamlessly with other Crawl4AI strategies:", "code_blocks": [{"language": "python", "code": "from crawl4ai import (\n    AsyncWebCrawler,\n    CrawlerRunConfig,\n    DefaultTableExtraction,\n    LLMExtractionStrategy,\n    JsonCssExtractionStrategy\n)\n\nasync def combined_extraction(url):\n    async…", "filename": ""}], "chunk_position": 53, "heading_path": "Combining with Other Strategies > Combining with Other Strategies", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Combining with Other Strategies > Combining with Other Strategies"}, {"id": "889090117448c6bc", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Optimization Tips", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Optimization Tips\n\n- **Disable when not needed** : Use `NoTableExtraction` if tables aren't required\n- **Target specific areas** : Use `css_selector` to limit processing scope\n- **Set minimum thresholds** : Filter out…", "code_blocks": [{"language": "python", "code": "# Optimized configuration for large pages\nconfig = CrawlerRunConfig(\n    # Only process main content area\n    css_selector=\"article.main-content\",\n\n    # Exclude navigation and sidebars…", "filename": ""}], "chunk_position": 53, "heading_path": "Optimization Tips > Optimization Tips", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Optimization Tips > Optimization Tips"}, {"id": "95e6591b416c8fb3", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Important: Your Code Still Works!", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Important: Your Code Still Works!\n\n**No changes required!** The transition to the strategy pattern is **fully backward compatible** .", "code_blocks": [], "chunk_position": 53, "heading_path": "Important: Your Code Still Works! > Important: Your Code Still Works!", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Important: Your Code Still Works! > Important: Your Code Still Works!"}, {"id": "1cec9f6bb3988267", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "v0.7.2 and Earlier", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: v0.7.2 and Earlier\n\n```python\n# Old way - directly passing table_score_threshold\nconfig = CrawlerRunConfig(\n    table_score_threshold=7\n)\n# Internally: No strategy pattern, direct implementation\n```", "code_blocks": [{"language": "python", "code": "# Old way - directly passing table_score_threshold\nconfig = CrawlerRunConfig(\n    table_score_threshold=7\n)\n# Internally: No strategy pattern, direct implementation", "filename": ""}], "chunk_position": 53, "heading_path": "v0.7.2 and Earlier > v0.7.2 and Earlier", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > v0.7.2 and Earlier > v0.7.2 and Earlier"}, {"id": "17b3c17b05f91a9f", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "v0.7.3+ (Current)", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: v0.7.3+ (Current)\n\n```python\n# Old way STILL WORKS - we handle it internally\nconfig = CrawlerRunConfig(\n    table_score_threshold=7\n)\n# Internally: Automatically creates DefaultTableExtraction(table_score_threshold=7)\n```", "code_blocks": [{"language": "python", "code": "# Old way STILL WORKS - we handle it internally\nconfig = CrawlerRunConfig(\n    table_score_threshold=7\n)\n# Internally: Automatically creates DefaultTableExtraction(table_score_threshold=7)", "filename": ""}], "chunk_position": 53, "heading_path": "v0.7.3+ (Current) > v0.7.3+ (Current)", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > v0.7.3+ (Current) > v0.7.3+ (Current)"}, {"id": "9890989505249947", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Taking Advantage of New Features", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Taking Advantage of New Features\n\nWhile your old code works, you can now use the strategy pattern for more control:", "code_blocks": [{"language": "python", "code": "# Option 1: Keep using the old way (perfectly fine!)\nconfig = CrawlerRunConfig(\n    table_score_threshold=7  # Still supported\n)\n\n# Option 2: Use the new strategy pattern (more flexibility)\nfrom…", "filename": ""}], "chunk_position": 53, "heading_path": "Taking Advantage of New Features > Taking Advantage of New Features", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Taking Advantage of New Features > Taking Advantage of New Features"}, {"id": "eb32b9e0fe94168a", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Summary", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Summary\n\n- ✅ **No breaking changes** - Old code works as-is\n- ✅ **Same defaults** - DefaultTableExtraction is automatically used\n- ✅ **Gradual adoption** - Use new features when you need them\n- ✅ **Full…", "code_blocks": [], "chunk_position": 53, "heading_path": "Summary > Summary", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Summary > Summary"}, {"id": "b18df30344cdf7c0", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "1. Choose the Right Strategy (Cost-Conscious Approach)", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: 1. Choose the Right Strategy (Cost-Conscious Approach)\n\n**Decision Flow** :\n\n**Strategy Selection Guide** :\n- **DefaultTableExtraction** : Use for 99% of cases - it's free and effective\n- **LLMTableExtraction** : Only for complex tables with merged cells…", "code_blocks": [{"language": "text", "code": "1. Do you need tables? \n   → No: Use NoTableExtraction\n   → Yes: Continue to #2\n\n2. Try DefaultTableExtraction first (FREE)\n   → Works? Done! ✅\n   → Fails? Continue to #3\n\n3. Is the table critical…", "filename": ""}], "chunk_position": 53, "heading_path": "1. Choose the Right Strategy (Cost-Conscious Approach) > 1. Choose the Right Strategy (Cost-Conscious Approach)", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > 1. Choose the Right Strategy (Cost-Conscious Approach) > 1. Choose the Right Strategy (Cost-Conscious Approach)"}, {"id": "e512aaab7e2d937e", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "2. Validate Extracted Data", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: 2. Validate Extracted Data\n\n```python\ndef validate_table(table):\n    \"\"\"Validate table data quality.\"\"\"\n    # Check structure\n    if not table.get('rows'):\n        return False\n\n    # Check consistency\n    if table.get('headers'):…\n```", "code_blocks": [{"language": "python", "code": "def validate_table(table):\n    \"\"\"Validate table data quality.\"\"\"\n    # Check structure\n    if not table.get('rows'):\n        return False\n\n    # Check consistency\n    if table.get('headers'):…", "filename": ""}], "chunk_position": 53, "heading_path": "2. Validate Extracted Data > 2. Validate Extracted Data", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > 2. Validate Extracted Data > 2. Validate Extracted Data"}, {"id": "7affc08dcea75f1c", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "3. Handle Edge Cases", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: 3. Handle Edge Cases\n\n```python\nasync def robust_table_extraction(url):\n    \"\"\"Extract tables with error handling.\"\"\"\n    async with AsyncWebCrawler() as crawler:\n        try:\n            config = CrawlerRunConfig(…\n```", "code_blocks": [{"language": "python", "code": "async def robust_table_extraction(url):\n    \"\"\"Extract tables with error handling.\"\"\"\n    async with AsyncWebCrawler() as crawler:\n        try:\n            config = CrawlerRunConfig(…", "filename": ""}], "chunk_position": 53, "heading_path": "3. Handle Edge Cases > 3. Handle Edge Cases", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > 3. Handle Edge Cases > 3. Handle Edge Cases"}, {"id": "14fd80310b84a18f", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Common Issues and Solutions", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Common Issues and Solutions\n\n| Issue | Cause | Solution |\n| --- | --- | --- |\n| No tables extracted | Score too high | Lower `table_score_threshold` |\n| Layout tables included | Score too low | Increase `table_score_threshold`…", "code_blocks": [], "chunk_position": 53, "heading_path": "Common Issues and Solutions > Common Issues and Solutions", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Common Issues and Solutions > Common Issues and Solutions"}, {"id": "e78198cfdd973cbf", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "Debug Logging", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Debug Logging\n\nEnable verbose logging to understand extraction decisions:", "code_blocks": [{"language": "python", "code": "import logging\n\n# Configure logging\nlogging.basicConfig(level=logging.DEBUG)\n\n# Enable verbose mode in strategy\nstrategy = DefaultTableExtraction(\n    table_score_threshold=7,\n    verbose=True  #…", "filename": ""}], "chunk_position": 53, "heading_path": "Debug Logging > Debug Logging", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > Debug Logging > Debug Logging"}, {"id": "c364241b00ffd6bd", "url": "https://docs.crawl4ai.com/core/table_extraction/", "page_title": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's table extraction strategies, including the default algorithm, LLM-based extraction, and custom strategies. It explains the strategy design pattern, configuration options,…", "heading": "See Also", "content": "Page: Table Extraction Strategies - Crawl4AI Documentation (v0.9.x)\nSection: See Also\n\n- [Extraction Strategies](extraction-strategies.md) - Overview of all extraction strategies\n- [Content Selection](../content-selection/) - Using CSS selectors and filters\n- [Performance…", "code_blocks": [], "chunk_position": 53, "heading_path": "See Also > See Also", "breadcrumbs": "Table Extraction Strategies - Crawl4AI Documentation (v0.9.x) > See Also > See Also"}, {"id": "3317ed0e11a8e661", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Why URL Seeding?", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Why URL Seeding?\n\nWeb crawling comes in different flavors, each with its own strengths. Let's understand when to use URL seeding versus deep crawling.", "code_blocks": [], "chunk_position": 54, "heading_path": "Why URL Seeding? > Why URL Seeding?", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Why URL Seeding? > Why URL Seeding?"}, {"id": "c99d01b3556663ad", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Deep Crawling: Real-Time Discovery", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Deep Crawling: Real-Time Discovery\n\nDeep crawling is perfect when you need:\n- **Fresh, real-time data** - discovering pages as they're created\n- **Dynamic exploration** - following links based on content\n- **Selective extraction** -…", "code_blocks": [{"language": "python", "code": "# Deep crawling example: Explore a website dynamically\nimport asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai.deep_crawling import BFSDeepCrawlStrategy\n\nasync def…", "filename": ""}], "chunk_position": 54, "heading_path": "Deep Crawling: Real-Time Discovery > Deep Crawling: Real-Time Discovery", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Deep Crawling: Real-Time Discovery > Deep Crawling: Real-Time Discovery"}, {"id": "d8c0a247bed3d116", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "URL Seeding: Bulk Discovery", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: URL Seeding: Bulk Discovery\n\nURL seeding shines when you want:\n- **Comprehensive coverage** - get thousands of URLs in seconds\n- **Bulk processing** - filter before crawling\n- **Resource efficiency** - know exactly what you'll…", "code_blocks": [{"language": "python", "code": "# URL seeding example: Analyze all documentation\nfrom crawl4ai import AsyncUrlSeeder, SeedingConfig\n\nseeder = AsyncUrlSeeder()\nconfig = SeedingConfig(\n    source=\"sitemap\",\n    extract_head=True,…", "filename": ""}], "chunk_position": 54, "heading_path": "URL Seeding: Bulk Discovery > URL Seeding: Bulk Discovery", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > URL Seeding: Bulk Discovery > URL Seeding: Bulk Discovery"}, {"id": "993e0a1bf2b70001", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "The Trade-offs", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: The Trade-offs\n\n| Aspect | Deep Crawling | URL Seeding |\n| --- | --- | --- |\n| **Coverage** | Discovers pages dynamically | Gets most existing URLs instantly |\n| **Freshness** | Finds brand new pages | May miss very…", "code_blocks": [], "chunk_position": 54, "heading_path": "The Trade-offs > The Trade-offs", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > The Trade-offs > The Trade-offs"}, {"id": "53d556bca971d93e", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "When to Use Each", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: When to Use Each\n\n**Choose Deep Crawling when:** \n- You need the absolute latest content\n- You're searching for specific information\n- The site structure is unknown or dynamic\n- You want to stop as soon as you find…", "code_blocks": [], "chunk_position": 54, "heading_path": "When to Use Each > When to Use Each", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > When to Use Each > When to Use Each"}, {"id": "c3aa6bfde50298d0", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Your First URL Seeding Adventure", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Your First URL Seeding Adventure\n\nLet's see the magic in action. We'll discover blog posts about Python, filter for tutorials, and crawl only those pages.\n\n**What just happened?** \n- We discovered all blog URLs from the sitemap+cc\n-…", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom crawl4ai import AsyncUrlSeeder, AsyncWebCrawler, SeedingConfig, CrawlerRunConfig\n\nasync def smart_blog_crawler():\n    # Step 1: Create our URL discoverer\n    seeder =…", "filename": ""}], "chunk_position": 54, "heading_path": "Your First URL Seeding Adventure > Your First URL Seeding Adventure", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Your First URL Seeding Adventure > Your First URL Seeding Adventure"}, {"id": "90bcb532ce4d0e3e", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Understanding the URL Seeder", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Understanding the URL Seeder\n\nNow that you've seen the magic, let's understand how it works.", "code_blocks": [], "chunk_position": 54, "heading_path": "Understanding the URL Seeder > Understanding the URL Seeder", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Understanding the URL Seeder > Understanding the URL Seeder"}, {"id": "52d87f9ca397322d", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Basic Usage", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Basic Usage\n\nCreating a URL seeder is simple:\n\nThe seeder can discover URLs from two powerful sources:", "code_blocks": [{"language": "python", "code": "from crawl4ai import AsyncUrlSeeder\n\n# Method 1: Manual cleanup\nseeder = AsyncUrlSeeder()\ntry:\n    config = SeedingConfig(source=\"sitemap\")\n    urls = await seeder.urls(\"example.com\",…", "filename": ""}], "chunk_position": 54, "heading_path": "Basic Usage > Basic Usage", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Basic Usage > Basic Usage"}, {"id": "5abc94201a65132a", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "1. Sitemaps (Fastest)", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: 1. Sitemaps (Fastest)\n\nSitemaps are XML files that websites create specifically to list all their URLs. It's like getting a menu at a restaurant - everything is listed upfront.\n\n**Sitemap Index Support**: For large…", "code_blocks": [{"language": "python", "code": "# Discover from sitemap\nconfig = SeedingConfig(source=\"sitemap\")\nurls = await seeder.urls(\"example.com\", config)", "filename": ""}, {"language": "xml", "code": "<!-- Example sitemap index -->\n<sitemapindex>\n  <sitemap>\n    <loc>https://techcrunch.com/sitemap-1.xml</loc>\n  </sitemap>\n  <sitemap>\n    <loc>https://techcrunch.com/sitemap-2.xml</loc>…", "filename": ""}], "chunk_position": 54, "heading_path": "1. Sitemaps (Fastest) > 1. Sitemaps (Fastest)", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > 1. Sitemaps (Fastest) > 1. Sitemaps (Fastest)"}, {"id": "aedc839795492e47", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "2. Common Crawl (Most Comprehensive)", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: 2. Common Crawl (Most Comprehensive)\n\nCommon Crawl is a massive public dataset that regularly crawls the entire web. It's like having access to a pre-built index of the internet.", "code_blocks": [{"language": "python", "code": "# Discover from Common Crawl\nconfig = SeedingConfig(source=\"cc\")\nurls = await seeder.urls(\"example.com\", config)", "filename": ""}], "chunk_position": 54, "heading_path": "2. Common Crawl (Most Comprehensive) > 2. Common Crawl (Most Comprehensive)", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > 2. Common Crawl (Most Comprehensive) > 2. Common Crawl (Most Comprehensive)"}, {"id": "b2bc50bbc93d4300", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "3. Both Sources (Maximum Coverage)", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: 3. Both Sources (Maximum Coverage)\n\n```python\n# Use both sources\nconfig = SeedingConfig(source=\"sitemap+cc\")\nurls = await seeder.urls(\"example.com\", config)\n```", "code_blocks": [{"language": "python", "code": "# Use both sources\nconfig = SeedingConfig(source=\"sitemap+cc\")\nurls = await seeder.urls(\"example.com\", config)", "filename": ""}], "chunk_position": 54, "heading_path": "3. Both Sources (Maximum Coverage) > 3. Both Sources (Maximum Coverage)", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > 3. Both Sources (Maximum Coverage) > 3. Both Sources (Maximum Coverage)"}, {"id": "5591a282845d015b", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Configuration Magic: SeedingConfig", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Configuration Magic: SeedingConfig\n\nThe `SeedingConfig` object is your control panel. Here's everything you can configure:", "code_blocks": [], "chunk_position": 54, "heading_path": "Configuration Magic: SeedingConfig > Configuration Magic: SeedingConfig", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Configuration Magic: SeedingConfig > Configuration Magic: SeedingConfig"}, {"id": "66d25da65b25b1af", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Pattern Matching Examples", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Pattern Matching Examples\n\n```python\n# Match all blog posts\nconfig = SeedingConfig(pattern=\"*/blog/*\")\n\n# Match only HTML files\nconfig = SeedingConfig(pattern=\"*.html\")\n\n# Match product pages\nconfig =…\n```", "code_blocks": [{"language": "python", "code": "# Match all blog posts\nconfig = SeedingConfig(pattern=\"*/blog/*\")\n\n# Match only HTML files\nconfig = SeedingConfig(pattern=\"*.html\")\n\n# Match product pages\nconfig =…", "filename": ""}], "chunk_position": 54, "heading_path": "Pattern Matching Examples > Pattern Matching Examples", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Pattern Matching Examples > Pattern Matching Examples"}, {"id": "dd9edf8ecffa74b3", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "URL Validation: Live Checking", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: URL Validation: Live Checking\n\nSometimes you need to know if URLs are actually accessible. That's where live checking comes in:\n\n**When to use live checking:** \n- Before a large crawling operation\n- When working with older…", "code_blocks": [{"language": "python", "code": "config = SeedingConfig(\n    source=\"sitemap\",\n    live_check=True,  # Verify each URL is accessible\n    concurrency=20    # Check 20 URLs in parallel\n)\nasync with AsyncUrlSeeder() as seeder:\n    urls…", "filename": ""}], "chunk_position": 54, "heading_path": "URL Validation: Live Checking > URL Validation: Live Checking", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > URL Validation: Live Checking > URL Validation: Live Checking"}, {"id": "9ec6975bf36d5dfd", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "The Power of Metadata: Head Extraction", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: The Power of Metadata: Head Extraction\n\nThis is where URL seeding gets really powerful. Instead of crawling entire pages, you can extract just the metadata:", "code_blocks": [{"language": "python", "code": "config = SeedingConfig(\n    extract_head=True  # Extract metadata from <head> section\n)\nasync with AsyncUrlSeeder() as seeder:\n    urls = await seeder.urls(\"example.com\", config)\n\n# Now each URL has…", "filename": ""}], "chunk_position": 54, "heading_path": "The Power of Metadata: Head Extraction > The Power of Metadata: Head Extraction", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > The Power of Metadata: Head Extraction > The Power of Metadata: Head Extraction"}, {"id": "601d94d56e7170f9", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "What Can We Extract?", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: What Can We Extract?\n\nThe head extraction gives you a treasure trove of information:", "code_blocks": [{"language": "python", "code": "# Example of extracted head_data\n{\n    \"title\": \"10 Python Tips for Beginners\",\n    \"charset\": \"utf-8\",\n    \"lang\": \"en\",\n    \"meta\": {\n        \"description\": \"Learn essential Python tips...\",…", "filename": ""}], "chunk_position": 54, "heading_path": "What Can We Extract? > What Can We Extract?", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > What Can We Extract? > What Can We Extract?"}, {"id": "5d2151dd690409f4", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Smart URL-Based Filtering (No Head Extraction)", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Smart URL-Based Filtering (No Head Extraction)\n\nWhen `extract_head=False` but you still provide a query, the seeder uses intelligent URL-based scoring:\n\nThis approach is much faster than head extraction while still providing intelligent filtering!", "code_blocks": [{"language": "python", "code": "# Fast filtering based on URL structure alone\nconfig = SeedingConfig(\n    source=\"sitemap\",\n    extract_head=False,  # Don't fetch page metadata\n    query=\"python tutorial async\",…", "filename": ""}], "chunk_position": 54, "heading_path": "Smart URL-Based Filtering (No Head Extraction) > Smart URL-Based Filtering (No Head Extraction)", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Smart URL-Based Filtering (No Head Extraction) > Smart URL-Based Filtering (No Head Extraction)"}, {"id": "56c3ed2a9adbee52", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Understanding Results", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Understanding Results\n\nEach URL in the results has this structure:\n\nLet's see a real example:", "code_blocks": [{"language": "python", "code": "{\n    \"url\": \"https://example.com/blog/python-tips.html\",\n    \"status\": \"valid\",        # \"valid\", \"not_valid\", or \"unknown\"\n    \"head_data\": {            # Only if extract_head=True\n        \"title\":…", "filename": ""}, {"language": "python", "code": "config = SeedingConfig(\n    source=\"sitemap\",\n    extract_head=True,\n    live_check=True\n)\nasync with AsyncUrlSeeder() as seeder:\n    urls = await seeder.urls(\"blog.example.com\", config)\n\n# Analyze…", "filename": ""}], "chunk_position": 54, "heading_path": "Understanding Results > Understanding Results", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Understanding Results > Understanding Results"}, {"id": "efd18812933f5903", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Smart Filtering with BM25 Scoring", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Smart Filtering with BM25 Scoring\n\nNow for the really cool part - intelligent filtering based on relevance!", "code_blocks": [], "chunk_position": 54, "heading_path": "Smart Filtering with BM25 Scoring > Smart Filtering with BM25 Scoring", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Smart Filtering with BM25 Scoring > Smart Filtering with BM25 Scoring"}, {"id": "d30696b9ffd16c18", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Introduction to Relevance Scoring", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Introduction to Relevance Scoring\n\nBM25 is a ranking algorithm that scores how relevant a document is to a search query. With URL seeding, we can score URLs based on their metadata *before* crawling them.\n\nThink of it like this:\n-…", "code_blocks": [], "chunk_position": 54, "heading_path": "Introduction to Relevance Scoring > Introduction to Relevance Scoring", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Introduction to Relevance Scoring > Introduction to Relevance Scoring"}, {"id": "22ed3201c3daf692", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Finding Documentation Pages", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Finding Documentation Pages\n\n```python\n# Find API documentation\nconfig = SeedingConfig(\n    source=\"sitemap\",\n    extract_head=True,\n    query=\"API reference documentation endpoints\",\n    scoring_method=\"bm25\",\n    score_threshold=0.5,…\n```", "code_blocks": [{"language": "python", "code": "# Find API documentation\nconfig = SeedingConfig(\n    source=\"sitemap\",\n    extract_head=True,\n    query=\"API reference documentation endpoints\",\n    scoring_method=\"bm25\",\n    score_threshold=0.5,…", "filename": ""}], "chunk_position": 54, "heading_path": "Finding Documentation Pages > Finding Documentation Pages", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Finding Documentation Pages > Finding Documentation Pages"}, {"id": "1cfb2c05ac36110c", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Discovering Product Pages", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Discovering Product Pages\n\n```python\n# Find specific products\nconfig = SeedingConfig(\n    source=\"sitemap+cc\",  # Use both sources\n    extract_head=True,\n    query=\"wireless headphones noise canceling\",\n    scoring_method=\"bm25\",…\n```", "code_blocks": [{"language": "python", "code": "# Find specific products\nconfig = SeedingConfig(\n    source=\"sitemap+cc\",  # Use both sources\n    extract_head=True,\n    query=\"wireless headphones noise canceling\",\n    scoring_method=\"bm25\",…", "filename": ""}], "chunk_position": 54, "heading_path": "Discovering Product Pages > Discovering Product Pages", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Discovering Product Pages > Discovering Product Pages"}, {"id": "b2f4b6900179c7a5", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Filtering News Articles", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Filtering News Articles\n\n```python\n# Find recent news about AI\nconfig = SeedingConfig(\n    source=\"sitemap\",\n    extract_head=True,\n    query=\"artificial intelligence machine learning breakthrough\",\n    scoring_method=\"bm25\",…\n```", "code_blocks": [{"language": "python", "code": "# Find recent news about AI\nconfig = SeedingConfig(\n    source=\"sitemap\",\n    extract_head=True,\n    query=\"artificial intelligence machine learning breakthrough\",\n    scoring_method=\"bm25\",…", "filename": ""}], "chunk_position": 54, "heading_path": "Filtering News Articles > Filtering News Articles", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Filtering News Articles > Filtering News Articles"}, {"id": "4746159e10fe01c7", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Complex Query Patterns", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Complex Query Patterns\n\n```python\n# Multi-concept queries\nqueries = [\n    \"python async await concurrency tutorial\",\n    \"data science pandas numpy visualization\",\n    \"web scraping beautifulsoup selenium automation\",\n    \"machine…\n```", "code_blocks": [{"language": "python", "code": "# Multi-concept queries\nqueries = [\n    \"python async await concurrency tutorial\",\n    \"data science pandas numpy visualization\",\n    \"web scraping beautifulsoup selenium automation\",\n    \"machine…", "filename": ""}], "chunk_position": 54, "heading_path": "Complex Query Patterns > Complex Query Patterns", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Complex Query Patterns > Complex Query Patterns"}, {"id": "aae35965be115ddb", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Scaling Up: Multiple Domains", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Scaling Up: Multiple Domains\n\nWhen you need to discover URLs across multiple websites, URL seeding really shines.", "code_blocks": [], "chunk_position": 54, "heading_path": "Scaling Up: Multiple Domains > Scaling Up: Multiple Domains", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Scaling Up: Multiple Domains > Scaling Up: Multiple Domains"}, {"id": "f087df52b3112a83", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "The `many_urls` Method", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: The `many_urls` Method\n\n```python\n# Discover URLs from multiple domains in parallel\ndomains = [\"site1.com\", \"site2.com\", \"site3.com\"]\n\nconfig = SeedingConfig(\n    source=\"sitemap\",\n    extract_head=True,\n    query=\"python tutorial\",…\n```", "code_blocks": [{"language": "python", "code": "# Discover URLs from multiple domains in parallel\ndomains = [\"site1.com\", \"site2.com\", \"site3.com\"]\n\nconfig = SeedingConfig(\n    source=\"sitemap\",\n    extract_head=True,\n    query=\"python tutorial\",…", "filename": ""}], "chunk_position": 54, "heading_path": "The `many_urls` Method > The `many_urls` Method", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > The `many_urls` Method > The `many_urls` Method"}, {"id": "b0a61f560acfb3ee", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Competitor Analysis", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Competitor Analysis\n\n```python\n# Analyze content strategies across competitors\ncompetitors = [\n    \"competitor1.com\",\n    \"competitor2.com\", \n    \"competitor3.com\"\n]\n\nconfig = SeedingConfig(\n    source=\"sitemap\",…\n```", "code_blocks": [{"language": "python", "code": "# Analyze content strategies across competitors\ncompetitors = [\n    \"competitor1.com\",\n    \"competitor2.com\", \n    \"competitor3.com\"\n]\n\nconfig = SeedingConfig(\n    source=\"sitemap\",…", "filename": ""}], "chunk_position": 54, "heading_path": "Competitor Analysis > Competitor Analysis", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Competitor Analysis > Competitor Analysis"}, {"id": "830b9bc61ae110f9", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Industry Research", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Industry Research\n\n```python\n# Research Python tutorials across educational sites\neducational_sites = [\n    \"realpython.com\",\n    \"pythontutorial.net\",\n    \"learnpython.org\",\n    \"python.org\"\n]\n\nconfig = SeedingConfig(…\n```", "code_blocks": [{"language": "python", "code": "# Research Python tutorials across educational sites\neducational_sites = [\n    \"realpython.com\",\n    \"pythontutorial.net\",\n    \"learnpython.org\",\n    \"python.org\"\n]\n\nconfig = SeedingConfig(…", "filename": ""}], "chunk_position": 54, "heading_path": "Industry Research > Industry Research", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Industry Research > Industry Research"}, {"id": "fb436966001e9f15", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Multi-Site Monitoring", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Multi-Site Monitoring\n\n```python\n# Monitor news about your company across multiple sources\nnews_sites = [\n    \"techcrunch.com\",\n    \"theverge.com\",\n    \"wired.com\",\n    \"arstechnica.com\"\n]\n\ncompany_name = \"YourCompany\"\n\nconfig =…\n```", "code_blocks": [{"language": "python", "code": "# Monitor news about your company across multiple sources\nnews_sites = [\n    \"techcrunch.com\",\n    \"theverge.com\",\n    \"wired.com\",\n    \"arstechnica.com\"\n]\n\ncompany_name = \"YourCompany\"\n\nconfig =…", "filename": ""}], "chunk_position": 54, "heading_path": "Multi-Site Monitoring > Multi-Site Monitoring", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Multi-Site Monitoring > Multi-Site Monitoring"}, {"id": "84e85b63a28573ca", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Advanced Integration Patterns", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Advanced Integration Patterns\n\nLet's put everything together in a real-world example.", "code_blocks": [], "chunk_position": 54, "heading_path": "Advanced Integration Patterns > Advanced Integration Patterns", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Advanced Integration Patterns > Advanced Integration Patterns"}, {"id": "24221b375313e545", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Building a Research Assistant", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Building a Research Assistant\n\nHere's a complete example that discovers, scores, filters, and crawls intelligently:", "code_blocks": [{"language": "python", "code": "import asyncio\nfrom datetime import datetime\nfrom crawl4ai import AsyncUrlSeeder, AsyncWebCrawler, SeedingConfig, CrawlerRunConfig\n\nclass ResearchAssistant:\n    def __init__(self):…", "filename": ""}], "chunk_position": 54, "heading_path": "Building a Research Assistant > Building a Research Assistant", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Building a Research Assistant > Building a Research Assistant"}, {"id": "d393e0e7c38a7328", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Performance Optimization Tips", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Performance Optimization Tips\n\n```python\n# First run - populate cache\nconfig = SeedingConfig(source=\"sitemap\", extract_head=True, force=True)\nurls = await seeder.urls(\"example.com\", config)\n\n# Subsequent runs - use cache (much…\n```\n\n```python\n# For many small requests (like HEAD checks)\nconfig = SeedingConfig(concurrency=50, hits_per_sec=20)\n\n# For fewer large requests (like full head extraction)\nconfig = SeedingConfig(concurrency=10,…\n```\n\n```python\n# When crawling many URLs\nasync with AsyncWebCrawler() as crawler:\n    # Assuming urls is a list of URL strings\n    crawl_results = await crawler.arun_many(urls, config=config)\n\n    # Process as they…\n```\n\n```python\n# Safe for domains with 1M+ URLs\nconfig = SeedingConfig(\n    source=\"cc+sitemap\",\n    concurrency=50,  # Queue size adapts to concurrency\n    max_urls=100000  # Process in batches if needed\n)\n\n# The…\n```", "code_blocks": [{"language": "python", "code": "# First run - populate cache\nconfig = SeedingConfig(source=\"sitemap\", extract_head=True, force=True)\nurls = await seeder.urls(\"example.com\", config)\n\n# Subsequent runs - use cache (much…", "filename": ""}, {"language": "python", "code": "# For many small requests (like HEAD checks)\nconfig = SeedingConfig(concurrency=50, hits_per_sec=20)\n\n# For fewer large requests (like full head extraction)\nconfig = SeedingConfig(concurrency=10,…", "filename": ""}, {"language": "python", "code": "# When crawling many URLs\nasync with AsyncWebCrawler() as crawler:\n    # Assuming urls is a list of URL strings\n    crawl_results = await crawler.arun_many(urls, config=config)\n\n    # Process as they…", "filename": ""}, {"language": "python", "code": "# Safe for domains with 1M+ URLs\nconfig = SeedingConfig(\n    source=\"cc+sitemap\",\n    concurrency=50,  # Queue size adapts to concurrency\n    max_urls=100000  # Process in batches if needed\n)\n\n# The…", "filename": ""}], "chunk_position": 54, "heading_path": "Performance Optimization Tips > Performance Optimization Tips", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Performance Optimization Tips > Performance Optimization Tips"}, {"id": "abda3f491fd26d94", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Cache Management", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Cache Management\n\nThe seeder automatically caches results to speed up repeated operations:", "code_blocks": [{"language": "text", "code": "- **Common Crawl cache** : `~/.crawl4ai/seeder_cache/[index]_[domain]_[hash].jsonl`\n- **Sitemap cache** : `~/.crawl4ai/seeder_cache/sitemap_[domain]_[hash].json`\n- **HEAD data cache** :…", "filename": ""}], "chunk_position": 54, "heading_path": "Cache Management > Cache Management", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Cache Management > Cache Management"}, {"id": "88d21ea4b5cb6f59", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Smart TTL Cache for Sitemaps", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Smart TTL Cache for Sitemaps\n\nSitemap caches now include intelligent validation:\n\n**Cache validation priority:** \n1. `force=True` → Always refetch\n2. Cache doesn't exist → Fetch fresh\n3. `validate_sitemap_lastmod=True` and…", "code_blocks": [{"language": "python", "code": "# Default: 24-hour TTL with lastmod validation\nconfig = SeedingConfig(\n    source=\"sitemap\",\n    cache_ttl_hours=24,              # Cache expires after 24 hours\n    validate_sitemap_lastmod=True    #…", "filename": ""}], "chunk_position": 54, "heading_path": "Smart TTL Cache for Sitemaps > Smart TTL Cache for Sitemaps", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Smart TTL Cache for Sitemaps > Smart TTL Cache for Sitemaps"}, {"id": "c9538a45d51bd54b", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Pattern Matching Strategies", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Pattern Matching Strategies\n\n```python\n# Be specific when possible\ngood_pattern = \"*/blog/2024/*.html\"  # Specific\nbad_pattern = \"*\"                     # Too broad\n\n# Combine patterns with metadata filtering\nconfig = SeedingConfig(…\n```", "code_blocks": [{"language": "python", "code": "# Be specific when possible\ngood_pattern = \"*/blog/2024/*.html\"  # Specific\nbad_pattern = \"*\"                     # Too broad\n\n# Combine patterns with metadata filtering\nconfig = SeedingConfig(…", "filename": ""}], "chunk_position": 54, "heading_path": "Pattern Matching Strategies > Pattern Matching Strategies", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Pattern Matching Strategies > Pattern Matching Strategies"}, {"id": "c94ef5335409b294", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Rate Limiting Considerations", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Rate Limiting Considerations\n\n```python\n# Be respectful of servers\nconfig = SeedingConfig(\n    hits_per_sec=10,      # Max 10 requests per second\n    concurrency=20        # But use 20 workers\n)\n\n# For your own servers\nconfig =…\n```", "code_blocks": [{"language": "python", "code": "# Be respectful of servers\nconfig = SeedingConfig(\n    hits_per_sec=10,      # Max 10 requests per second\n    concurrency=20        # But use 20 workers\n)\n\n# For your own servers\nconfig =…", "filename": ""}], "chunk_position": 54, "heading_path": "Rate Limiting Considerations > Rate Limiting Considerations", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Rate Limiting Considerations > Rate Limiting Considerations"}, {"id": "a7f891620ecc2ad5", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Common Patterns", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Common Patterns\n\n```python\n# Blog post discovery\nconfig = SeedingConfig(\n    source=\"sitemap\",\n    pattern=\"*/blog/*\",\n    extract_head=True,\n    query=\"your topic\",\n    scoring_method=\"bm25\"\n)\n\n# E-commerce product…\n```", "code_blocks": [{"language": "python", "code": "# Blog post discovery\nconfig = SeedingConfig(\n    source=\"sitemap\",\n    pattern=\"*/blog/*\",\n    extract_head=True,\n    query=\"your topic\",\n    scoring_method=\"bm25\"\n)\n\n# E-commerce product…", "filename": ""}], "chunk_position": 54, "heading_path": "Common Patterns > Common Patterns", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Common Patterns > Common Patterns"}, {"id": "4d9f91bcef2b2031", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Troubleshooting Guide", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Troubleshooting Guide\n\n| Issue | Solution |\n| --- | --- |\n| No URLs found | Try `source=\"cc+sitemap\"`, check domain spelling |\n| Slow discovery | Reduce `concurrency`, add `hits_per_sec` limit |\n| Missing metadata | Ensure…", "code_blocks": [], "chunk_position": 54, "heading_path": "Troubleshooting Guide > Troubleshooting Guide", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Troubleshooting Guide > Troubleshooting Guide"}, {"id": "f575ce59e0e89a53", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Performance Benchmarks", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Performance Benchmarks\n\nTypical performance on a standard connection:\n- **Sitemap discovery** : 100-1,000 URLs/second\n- **Common Crawl discovery** : 50-500 URLs/second\n- **HEAD checking** : 10-50 URLs/second\n- **Head…", "code_blocks": [], "chunk_position": 54, "heading_path": "Performance Benchmarks > Performance Benchmarks", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Performance Benchmarks > Performance Benchmarks"}, {"id": "4472e7832174268d", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Conclusion", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Conclusion\n\nURL seeding transforms web crawling from a blind expedition into a surgical strike. By discovering and analyzing URLs before crawling, you can:\n- Save hours of crawling time\n- Reduce bandwidth usage…", "code_blocks": [], "chunk_position": 54, "heading_path": "Conclusion > Conclusion", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Conclusion > Conclusion"}, {"id": "d8a05a169bd713bd", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Smart URL Filtering", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Smart URL Filtering\n\nThe seeder automatically filters out nonsense URLs that aren't useful for content crawling:\n\nTo disable filtering (not recommended):", "code_blocks": [{"language": "python", "code": "# Enabled by default\nconfig = SeedingConfig(\n    source=\"sitemap\",\n    filter_nonsense_urls=True  # Default: True\n)\n\n# URLs that get filtered:\n# - robots.txt, sitemap.xml, ads.txt\n# - API endpoints…", "filename": ""}, {"language": "python", "code": "config = SeedingConfig(\n    source=\"sitemap\",\n    filter_nonsense_urls=False  # Include ALL URLs\n)", "filename": ""}], "chunk_position": 54, "heading_path": "Smart URL Filtering > Smart URL Filtering", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Smart URL Filtering > Smart URL Filtering"}, {"id": "40f43c118811a498", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Key Features Summary", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Key Features Summary\n\n- **Parallel Sitemap Index Processing** : Automatically detects and processes sitemap indexes in parallel\n- **Memory Protection** : Bounded queues prevent RAM issues with large domains (1M+ URLs)\n-…", "code_blocks": [], "chunk_position": 54, "heading_path": "Key Features Summary > Key Features Summary", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Key Features Summary > Key Features Summary"}, {"id": "f8eb581015ea7e50", "url": "https://docs.crawl4ai.com/core/url-seeding/", "page_title": "URL Seeding: The Smart Way to Crawl at Scale", "page_type": "guide", "page_summary": "This page explains how to use URL seeding to discover and filter URLs before crawling, covering configuration, smart filtering with BM25 scoring, and scaling across multiple domains.", "heading": "Need More Coverage?", "content": "Page: URL Seeding: The Smart Way to Crawl at Scale\nSection: Need More Coverage?\n\nIf you need to discover URLs across an entire domain — including subdomains, hidden services, and pages not listed in any sitemap — check out [Domain Mapping](../domain-mapping/). It combines 8…", "code_blocks": [], "chunk_position": 54, "heading_path": "Need More Coverage? > Need More Coverage?", "breadcrumbs": "URL Seeding: The Smart Way to Crawl at Scale > Need More Coverage? > Need More Coverage?"}, {"id": "3a0b29b7fbec00c4", "url": "https://docs.crawl4ai.com/extraction/chunking/", "page_title": "Chunking Strategies", "page_type": "guide", "page_summary": "This page explains various chunking strategies for dividing large texts into manageable parts, including regex-based, sentence-based, topic-based, fixed-length word, and sliding window chunking, as…", "heading": "Chunking Strategies", "content": "Page: Chunking Strategies\nSection: Chunking Strategies\n\nChunking strategies are critical for dividing large texts into manageable parts, enabling effective content processing and extraction. These strategies are foundational in cosine similarity-based…", "code_blocks": [], "chunk_position": 55, "heading_path": "Chunking Strategies > Chunking Strategies", "breadcrumbs": "Chunking Strategies > Chunking Strategies > Chunking Strategies"}, {"id": "b398e15975abd4e1", "url": "https://docs.crawl4ai.com/extraction/chunking/", "page_title": "Chunking Strategies", "page_type": "guide", "page_summary": "This page explains various chunking strategies for dividing large texts into manageable parts, including regex-based, sentence-based, topic-based, fixed-length word, and sliding window chunking, as…", "heading": "Why Use Chunking?", "content": "Page: Chunking Strategies\nSection: Why Use Chunking?\n\n1. **Cosine Similarity and Query Relevance**: Prepares chunks for semantic similarity analysis.\n2. **RAG System Integration**: Seamlessly processes and stores chunks for retrieval.\n3. **Structured…", "code_blocks": [], "chunk_position": 55, "heading_path": "Why Use Chunking? > Why Use Chunking?", "breadcrumbs": "Chunking Strategies > Why Use Chunking? > Why Use Chunking?"}, {"id": "41e015d7b0ae180a", "url": "https://docs.crawl4ai.com/extraction/chunking/", "page_title": "Chunking Strategies", "page_type": "guide", "page_summary": "This page explains various chunking strategies for dividing large texts into manageable parts, including regex-based, sentence-based, topic-based, fixed-length word, and sliding window chunking, as…", "heading": "1. Regex-Based Chunking", "content": "Page: Chunking Strategies\nSection: 1. Regex-Based Chunking\n\nSplits text based on regular expression patterns, useful for coarse segmentation.\n\n**Code Example**:", "code_blocks": [{"language": "python", "code": "class RegexChunking:\n    def __init__(self, patterns=None):\n        self.patterns = patterns or [r'\\n\\n']  # Default pattern for paragraphs\n\n    def chunk(self, text):\n        paragraphs = [text]…", "filename": ""}], "chunk_position": 55, "heading_path": "1. Regex-Based Chunking > 1. Regex-Based Chunking", "breadcrumbs": "Chunking Strategies > 1. Regex-Based Chunking > 1. Regex-Based Chunking"}, {"id": "df8b0e066f2a5beb", "url": "https://docs.crawl4ai.com/extraction/chunking/", "page_title": "Chunking Strategies", "page_type": "guide", "page_summary": "This page explains various chunking strategies for dividing large texts into manageable parts, including regex-based, sentence-based, topic-based, fixed-length word, and sliding window chunking, as…", "heading": "2. Sentence-Based Chunking", "content": "Page: Chunking Strategies\nSection: 2. Sentence-Based Chunking\n\nDivides text into sentences using NLP tools, ideal for extracting meaningful statements.\n\n**Code Example**:", "code_blocks": [{"language": "python", "code": "from nltk.tokenize import sent_tokenize\n\nclass NlpSentenceChunking:\n    def chunk(self, text):\n        sentences = sent_tokenize(text)\n        return [sentence.strip() for sentence in sentences]\n\n#…", "filename": ""}], "chunk_position": 55, "heading_path": "2. Sentence-Based Chunking > 2. Sentence-Based Chunking", "breadcrumbs": "Chunking Strategies > 2. Sentence-Based Chunking > 2. Sentence-Based Chunking"}, {"id": "2f7ff9688187eff1", "url": "https://docs.crawl4ai.com/extraction/chunking/", "page_title": "Chunking Strategies", "page_type": "guide", "page_summary": "This page explains various chunking strategies for dividing large texts into manageable parts, including regex-based, sentence-based, topic-based, fixed-length word, and sliding window chunking, as…", "heading": "3. Topic-Based Segmentation", "content": "Page: Chunking Strategies\nSection: 3. Topic-Based Segmentation\n\nUses algorithms like TextTiling to create topic-coherent chunks.\n\n**Code Example**:", "code_blocks": [{"language": "python", "code": "from nltk.tokenize import TextTilingTokenizer\n\nclass TopicSegmentationChunking:\n    def __init__(self):\n        self.tokenizer = TextTilingTokenizer()\n\n    def chunk(self, text):\n        return…", "filename": ""}], "chunk_position": 55, "heading_path": "3. Topic-Based Segmentation > 3. Topic-Based Segmentation", "breadcrumbs": "Chunking Strategies > 3. Topic-Based Segmentation > 3. Topic-Based Segmentation"}, {"id": "1ec74690c1fc9c1a", "url": "https://docs.crawl4ai.com/extraction/chunking/", "page_title": "Chunking Strategies", "page_type": "guide", "page_summary": "This page explains various chunking strategies for dividing large texts into manageable parts, including regex-based, sentence-based, topic-based, fixed-length word, and sliding window chunking, as…", "heading": "4. Fixed-Length Word Chunking", "content": "Page: Chunking Strategies\nSection: 4. Fixed-Length Word Chunking\n\nSegments text into chunks of a fixed word count.\n\n**Code Example**:", "code_blocks": [{"language": "python", "code": "class FixedLengthWordChunking:\n    def __init__(self, chunk_size=100):\n        self.chunk_size = chunk_size\n\n    def chunk(self, text):\n        words = text.split()\n        return [' '.join(words[i:i…", "filename": ""}], "chunk_position": 55, "heading_path": "4. Fixed-Length Word Chunking > 4. Fixed-Length Word Chunking", "breadcrumbs": "Chunking Strategies > 4. Fixed-Length Word Chunking > 4. Fixed-Length Word Chunking"}, {"id": "da297e0ab3efa9be", "url": "https://docs.crawl4ai.com/extraction/chunking/", "page_title": "Chunking Strategies", "page_type": "guide", "page_summary": "This page explains various chunking strategies for dividing large texts into manageable parts, including regex-based, sentence-based, topic-based, fixed-length word, and sliding window chunking, as…", "heading": "5. Sliding Window Chunking", "content": "Page: Chunking Strategies\nSection: 5. Sliding Window Chunking\n\nGenerates overlapping chunks for better contextual coherence.\n\n**Code Example**:", "code_blocks": [{"language": "python", "code": "class SlidingWindowChunking:\n    def __init__(self, window_size=100, step=50):\n        self.window_size = window_size\n        self.step = step\n\n    def chunk(self, text):\n        words =…", "filename": ""}], "chunk_position": 55, "heading_path": "5. Sliding Window Chunking > 5. Sliding Window Chunking", "breadcrumbs": "Chunking Strategies > 5. Sliding Window Chunking > 5. Sliding Window Chunking"}, {"id": "9f1ad055ad3c531b", "url": "https://docs.crawl4ai.com/extraction/chunking/", "page_title": "Chunking Strategies", "page_type": "guide", "page_summary": "This page explains various chunking strategies for dividing large texts into manageable parts, including regex-based, sentence-based, topic-based, fixed-length word, and sliding window chunking, as…", "heading": "Combining Chunking with Cosine Similarity", "content": "Page: Chunking Strategies\nSection: Combining Chunking with Cosine Similarity\n\nTo enhance the relevance of extracted content, chunking strategies can be paired with cosine similarity techniques. Here’s an example workflow:\n\n**Code Example**:", "code_blocks": [{"language": "python", "code": "from sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.metrics.pairwise import cosine_similarity\n\nclass CosineSimilarityExtractor:\n    def __init__(self, query):\n        self.query…", "filename": ""}], "chunk_position": 55, "heading_path": "Combining Chunking with Cosine Similarity > Combining Chunking with Cosine Similarity", "breadcrumbs": "Chunking Strategies > Combining Chunking with Cosine Similarity > Combining Chunking with Cosine Similarity"}, {"id": "c3eff4d378b18cde", "url": "https://docs.crawl4ai.com/extraction/clustring-strategies/", "page_title": "Clustering Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Cosine Strategy in Crawl4AI for semantic content extraction, covering configuration options, usage examples, best practices, and error handling.", "heading": "Cosine Strategy", "content": "Page: Clustering Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Cosine Strategy\n\nThe Cosine Strategy in Crawl4AI uses similarity-based clustering to identify and extract relevant content sections from web pages. This strategy is particularly useful when you need to find and…", "code_blocks": [], "chunk_position": 56, "heading_path": "Cosine Strategy > Cosine Strategy", "breadcrumbs": "Clustering Strategies - Crawl4AI Documentation (v0.9.x) > Cosine Strategy > Cosine Strategy"}, {"id": "45427e08dc259f1c", "url": "https://docs.crawl4ai.com/extraction/clustring-strategies/", "page_title": "Clustering Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Cosine Strategy in Crawl4AI for semantic content extraction, covering configuration options, usage examples, best practices, and error handling.", "heading": "How It Works", "content": "Page: Clustering Strategies - Crawl4AI Documentation (v0.9.x)\nSection: How It Works\n\nThe Cosine Strategy:\n1. Breaks down page content into meaningful chunks\n2. Converts text into vector representations\n3. Calculates similarity between chunks\n4. Clusters similar content together\n5.…", "code_blocks": [], "chunk_position": 56, "heading_path": "How It Works > How It Works", "breadcrumbs": "Clustering Strategies - Crawl4AI Documentation (v0.9.x) > How It Works > How It Works"}, {"id": "5aa2132740aeb4bf", "url": "https://docs.crawl4ai.com/extraction/clustring-strategies/", "page_title": "Clustering Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Cosine Strategy in Crawl4AI for semantic content extraction, covering configuration options, usage examples, best practices, and error handling.", "heading": "Basic Usage", "content": "Page: Clustering Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Basic Usage\n\n```python\nfrom crawl4ai import CosineStrategy\n\nstrategy = CosineStrategy(\n    semantic_filter=\"product reviews\",    # Target content type\n    word_count_threshold=10,             # Minimum words per cluster…\n```", "code_blocks": [{"language": "python", "code": "from crawl4ai import CosineStrategy\n\nstrategy = CosineStrategy(\n    semantic_filter=\"product reviews\",    # Target content type\n    word_count_threshold=10,             # Minimum words per cluster…", "filename": ""}], "chunk_position": 56, "heading_path": "Basic Usage > Basic Usage", "breadcrumbs": "Clustering Strategies - Crawl4AI Documentation (v0.9.x) > Basic Usage > Basic Usage"}, {"id": "67acb5ee5faf4f4b", "url": "https://docs.crawl4ai.com/extraction/clustring-strategies/", "page_title": "Clustering Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Cosine Strategy in Crawl4AI for semantic content extraction, covering configuration options, usage examples, best practices, and error handling.", "heading": "Core Parameters", "content": "Page: Clustering Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Core Parameters\n\n```python\nCosineStrategy(\n    # Content Filtering\n    semantic_filter: str = None,       # Keywords/topic for content filtering\n    word_count_threshold: int = 10,    # Minimum words per cluster…\n```", "code_blocks": [{"language": "python", "code": "CosineStrategy(\n    # Content Filtering\n    semantic_filter: str = None,       # Keywords/topic for content filtering\n    word_count_threshold: int = 10,    # Minimum words per cluster…", "filename": ""}], "chunk_position": 56, "heading_path": "Core Parameters > Core Parameters", "breadcrumbs": "Clustering Strategies - Crawl4AI Documentation (v0.9.x) > Core Parameters > Core Parameters"}, {"id": "1b23bbcbafa53068", "url": "https://docs.crawl4ai.com/extraction/clustring-strategies/", "page_title": "Clustering Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Cosine Strategy in Crawl4AI for semantic content extraction, covering configuration options, usage examples, best practices, and error handling.", "heading": "Parameter Details", "content": "Page: Clustering Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Parameter Details\n\n1. **semantic_filter** \n   - Sets the target topic or content type\n   - Use keywords relevant to your desired content\n   - Example: \"technical specifications\", \"user reviews\", \"pricing…", "code_blocks": [{"language": "python", "code": "# Strict matching\nstrategy = CosineStrategy(sim_threshold=0.8)\n\n# Loose matching\nstrategy = CosineStrategy(sim_threshold=0.3)", "filename": ""}, {"language": "python", "code": "# Only consider substantial paragraphs\nstrategy = CosineStrategy(word_count_threshold=50)", "filename": ""}, {"language": "python", "code": "# Get top 5 most relevant content clusters\nstrategy = CosineStrategy(top_k=5)", "filename": ""}], "chunk_position": 56, "heading_path": "Parameter Details > Parameter Details", "breadcrumbs": "Clustering Strategies - Crawl4AI Documentation (v0.9.x) > Parameter Details > Parameter Details"}, {"id": "0afd12d4b77888c7", "url": "https://docs.crawl4ai.com/extraction/clustring-strategies/", "page_title": "Clustering Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Cosine Strategy in Crawl4AI for semantic content extraction, covering configuration options, usage examples, best practices, and error handling.", "heading": "1. Article Content Extraction", "content": "Page: Clustering Strategies - Crawl4AI Documentation (v0.9.x)\nSection: 1. Article Content Extraction\n\n```python\nstrategy = CosineStrategy(\n    semantic_filter=\"main article content\",\n    word_count_threshold=100,  # Longer blocks for articles\n    top_k=1                   # Usually want single main…\n```", "code_blocks": [{"language": "python", "code": "strategy = CosineStrategy(\n    semantic_filter=\"main article content\",\n    word_count_threshold=100,  # Longer blocks for articles\n    top_k=1                   # Usually want single main…", "filename": ""}], "chunk_position": 56, "heading_path": "1. Article Content Extraction > 1. Article Content Extraction", "breadcrumbs": "Clustering Strategies - Crawl4AI Documentation (v0.9.x) > 1. Article Content Extraction > 1. Article Content Extraction"}, {"id": "0e02dd3fb224691d", "url": "https://docs.crawl4ai.com/extraction/clustring-strategies/", "page_title": "Clustering Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Cosine Strategy in Crawl4AI for semantic content extraction, covering configuration options, usage examples, best practices, and error handling.", "heading": "2. Product Review Analysis", "content": "Page: Clustering Strategies - Crawl4AI Documentation (v0.9.x)\nSection: 2. Product Review Analysis\n\n```python\nstrategy = CosineStrategy(\n    semantic_filter=\"customer reviews and ratings\",\n    word_count_threshold=20,   # Reviews can be shorter\n    top_k=10,                 # Get multiple reviews…\n```", "code_blocks": [{"language": "python", "code": "strategy = CosineStrategy(\n    semantic_filter=\"customer reviews and ratings\",\n    word_count_threshold=20,   # Reviews can be shorter\n    top_k=10,                 # Get multiple reviews…", "filename": ""}], "chunk_position": 56, "heading_path": "2. Product Review Analysis > 2. Product Review Analysis", "breadcrumbs": "Clustering Strategies - Crawl4AI Documentation (v0.9.x) > 2. Product Review Analysis > 2. Product Review Analysis"}, {"id": "5e50c87e7a5cb22e", "url": "https://docs.crawl4ai.com/extraction/clustring-strategies/", "page_title": "Clustering Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Cosine Strategy in Crawl4AI for semantic content extraction, covering configuration options, usage examples, best practices, and error handling.", "heading": "3. Technical Documentation", "content": "Page: Clustering Strategies - Crawl4AI Documentation (v0.9.x)\nSection: 3. Technical Documentation\n\n```python\nstrategy = CosineStrategy(\n    semantic_filter=\"technical specifications documentation\",\n    word_count_threshold=30,\n    sim_threshold=0.6,        # Stricter matching for technical content…\n```", "code_blocks": [{"language": "python", "code": "strategy = CosineStrategy(\n    semantic_filter=\"technical specifications documentation\",\n    word_count_threshold=30,\n    sim_threshold=0.6,        # Stricter matching for technical content…", "filename": ""}], "chunk_position": 56, "heading_path": "3. Technical Documentation > 3. Technical Documentation", "breadcrumbs": "Clustering Strategies - Crawl4AI Documentation (v0.9.x) > 3. Technical Documentation > 3. Technical Documentation"}, {"id": "a83b145869e674ec", "url": "https://docs.crawl4ai.com/extraction/clustring-strategies/", "page_title": "Clustering Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Cosine Strategy in Crawl4AI for semantic content extraction, covering configuration options, usage examples, best practices, and error handling.", "heading": "Custom Clustering", "content": "Page: Clustering Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Custom Clustering\n\n```python\nstrategy = CosineStrategy(\n    linkage_method='complete',  # Alternative clustering method\n    max_dist=0.4,              # Larger clusters…\n```", "code_blocks": [{"language": "python", "code": "strategy = CosineStrategy(\n    linkage_method='complete',  # Alternative clustering method\n    max_dist=0.4,              # Larger clusters…", "filename": ""}], "chunk_position": 56, "heading_path": "Custom Clustering > Custom Clustering", "breadcrumbs": "Clustering Strategies - Crawl4AI Documentation (v0.9.x) > Custom Clustering > Custom Clustering"}, {"id": "23f245c87e6294f3", "url": "https://docs.crawl4ai.com/extraction/clustring-strategies/", "page_title": "Clustering Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Cosine Strategy in Crawl4AI for semantic content extraction, covering configuration options, usage examples, best practices, and error handling.", "heading": "Content Filtering Pipeline", "content": "Page: Clustering Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Content Filtering Pipeline\n\n```python\nstrategy = CosineStrategy(\n    semantic_filter=\"pricing plans features\",\n    word_count_threshold=15,\n    sim_threshold=0.5,\n    top_k=3\n)\n\nasync def extract_pricing_features(url: str):\n    async…\n```", "code_blocks": [{"language": "python", "code": "strategy = CosineStrategy(\n    semantic_filter=\"pricing plans features\",\n    word_count_threshold=15,\n    sim_threshold=0.5,\n    top_k=3\n)\n\nasync def extract_pricing_features(url: str):\n    async…", "filename": ""}], "chunk_position": 56, "heading_path": "Content Filtering Pipeline > Content Filtering Pipeline", "breadcrumbs": "Clustering Strategies - Crawl4AI Documentation (v0.9.x) > Content Filtering Pipeline > Content Filtering Pipeline"}, {"id": "c55300148993fd80", "url": "https://docs.crawl4ai.com/extraction/clustring-strategies/", "page_title": "Clustering Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Cosine Strategy in Crawl4AI for semantic content extraction, covering configuration options, usage examples, best practices, and error handling.", "heading": "Best Practices", "content": "Page: Clustering Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Best Practices\n\n1. **Adjust Thresholds Iteratively** \n   - Start with default values\n   - Adjust based on results\n   - Monitor clustering quality\n\n2. **Choose Appropriate Word Count Thresholds** \n   - Higher for…", "code_blocks": [{"language": "python", "code": "strategy = CosineStrategy(\n    word_count_threshold=10,  # Filter early\n    top_k=5,                 # Limit results\n    verbose=True             # Monitor performance\n)", "filename": ""}, {"language": "python", "code": "# For mixed content pages\nstrategy = CosineStrategy(\n    semantic_filter=\"product features\",\n    sim_threshold=0.4,      # More flexible matching\n    max_dist=0.3,          # Larger clusters…", "filename": ""}], "chunk_position": 56, "heading_path": "Best Practices > Best Practices", "breadcrumbs": "Clustering Strategies - Crawl4AI Documentation (v0.9.x) > Best Practices > Best Practices"}, {"id": "7eb618bd0810fb35", "url": "https://docs.crawl4ai.com/extraction/clustring-strategies/", "page_title": "Clustering Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Cosine Strategy in Crawl4AI for semantic content extraction, covering configuration options, usage examples, best practices, and error handling.", "heading": "Error Handling", "content": "Page: Clustering Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Error Handling\n\n```python\ntry:\n    result = await crawler.arun(\n        url=\"https://example.com\",\n        extraction_strategy=strategy\n    )\n\n    if result.success:\n        content = json.loads(result.extracted_content)…\n```", "code_blocks": [{"language": "python", "code": "try:\n    result = await crawler.arun(\n        url=\"https://example.com\",\n        extraction_strategy=strategy\n    )\n\n    if result.success:\n        content = json.loads(result.extracted_content)…", "filename": ""}], "chunk_position": 56, "heading_path": "Error Handling > Error Handling", "breadcrumbs": "Clustering Strategies - Crawl4AI Documentation (v0.9.x) > Error Handling > Error Handling"}, {"id": "88d826e19c2f7037", "url": "https://docs.crawl4ai.com/extraction/clustring-strategies/", "page_title": "Clustering Strategies - Crawl4AI Documentation (v0.9.x)", "page_type": "guide", "page_summary": "This page provides a comprehensive guide to using the Cosine Strategy in Crawl4AI for semantic content extraction, covering configuration options, usage examples, best practices, and error handling.", "heading": "Conclusion", "content": "Page: Clustering Strategies - Crawl4AI Documentation (v0.9.x)\nSection: Conclusion\n\nThe Cosine Strategy is particularly effective when:\n- Content structure is inconsistent\n- You need semantic understanding\n- You want to find similar content blocks\n- Structure-based extraction…", "code_blocks": [], "chunk_position": 56, "heading_path": "Conclusion > Conclusion", "breadcrumbs": "Clustering Strategies - Crawl4AI Documentation (v0.9.x) > Conclusion > Conclusion"}, {"id": "72dea9570ce35e2d", "url": "https://docs.crawl4ai.com/extraction/llm-strategies/", "page_title": "Extracting JSON (LLM)", "page_type": "guide", "page_summary": "A guide to using Crawl4AI's LLM-based extraction strategy to extract structured JSON from web pages using any LLM via LiteLLM, including schema definition, chunking, input formats, and practical…", "heading": "Overview", "content": "Page: Extracting JSON (LLM)\nSection: Overview\n\nIn some cases, you need to extract **complex or unstructured** information from a webpage that a simple CSS/XPath schema cannot easily parse. Or you want **AI**-driven insights, classification, or…", "code_blocks": [], "chunk_position": 57, "heading_path": "Overview > Overview", "breadcrumbs": "Extracting JSON (LLM) > Overview > Overview"}, {"id": "75028b523b1aa95d", "url": "https://docs.crawl4ai.com/extraction/llm-strategies/", "page_title": "Extracting JSON (LLM)", "page_type": "guide", "page_summary": "A guide to using Crawl4AI's LLM-based extraction strategy to extract structured JSON from web pages using any LLM via LiteLLM, including schema definition, chunking, input formats, and practical…", "heading": "1. Why Use an LLM?", "content": "Page: Extracting JSON (LLM)\nSection: 1. Why Use an LLM?\n\n- **Complex Reasoning**: If the site’s data is unstructured, scattered, or full of natural language context.\n- **Semantic Extraction**: Summaries, knowledge graphs, or relational data that require…", "code_blocks": [], "chunk_position": 57, "heading_path": "1. Why Use an LLM? > 1. Why Use an LLM?", "breadcrumbs": "Extracting JSON (LLM) > 1. Why Use an LLM? > 1. Why Use an LLM?"}, {"id": "0532bc54ea9ef1f0", "url": "https://docs.crawl4ai.com/extraction/llm-strategies/", "page_title": "Extracting JSON (LLM)", "page_type": "guide", "page_summary": "A guide to using Crawl4AI's LLM-based extraction strategy to extract structured JSON from web pages using any LLM via LiteLLM, including schema definition, chunking, input formats, and practical…", "heading": "2. Provider-Agnostic via LiteLLM", "content": "Page: Extracting JSON (LLM)\nSection: 2. Provider-Agnostic via LiteLLM\n\nYou can use LLMConfig, to quickly configure multiple variations of LLMs and experiment with them to find the optimal one for your use case. You can read more about LLMConfig…", "code_blocks": [{"language": "python", "code": "llm_config = LLMConfig(provider=\"openai/gpt-4o-mini\", api_token=os.getenv(\"OPENAI_API_KEY\"))", "filename": ""}], "chunk_position": 57, "heading_path": "2. Provider-Agnostic via LiteLLM > 2. Provider-Agnostic via LiteLLM", "breadcrumbs": "Extracting JSON (LLM) > 2. Provider-Agnostic via LiteLLM > 2. Provider-Agnostic via LiteLLM"}, {"id": "d6be9a7b26068b84", "url": "https://docs.crawl4ai.com/extraction/llm-strategies/", "page_title": "Extracting JSON (LLM)", "page_type": "guide", "page_summary": "A guide to using Crawl4AI's LLM-based extraction strategy to extract structured JSON from web pages using any LLM via LiteLLM, including schema definition, chunking, input formats, and practical…", "heading": "3.1 Flow", "content": "Page: Extracting JSON (LLM)\nSection: 3.1 Flow\n\n1. **Chunking** (optional): The HTML or markdown is split into smaller segments if it’s very long (based on `chunk_token_threshold`, overlap, etc.).\n2. **Prompt Construction**: For each chunk, the…", "code_blocks": [], "chunk_position": 57, "heading_path": "3.1 Flow > 3.1 Flow", "breadcrumbs": "Extracting JSON (LLM) > 3.1 Flow > 3.1 Flow"}, {"id": "8ad1d0eb26cc6e14", "url": "https://docs.crawl4ai.com/extraction/llm-strategies/", "page_title": "Extracting JSON (LLM)", "page_type": "guide", "page_summary": "A guide to using Crawl4AI's LLM-based extraction strategy to extract structured JSON from web pages using any LLM via LiteLLM, including schema definition, chunking, input formats, and practical…", "heading": "3.2 `extraction_type`", "content": "Page: Extracting JSON (LLM)\nSection: 3.2 `extraction_type`\n\n- **`\"schema\"`**: The model tries to return JSON conforming to your Pydantic-based schema.\n- **`\"block\"`**: The model returns freeform text, or smaller JSON structures, which the library…", "code_blocks": [], "chunk_position": 57, "heading_path": "3.2 `extraction_type` > 3.2 `extraction_type`", "breadcrumbs": "Extracting JSON (LLM) > 3.2 `extraction_type` > 3.2 `extraction_type`"}, {"id": "5501825249a78555", "url": "https://docs.crawl4ai.com/extraction/llm-strategies/", "page_title": "Extracting JSON (LLM)", "page_type": "guide", "page_summary": "A guide to using Crawl4AI's LLM-based extraction strategy to extract structured JSON from web pages using any LLM via LiteLLM, including schema definition, chunking, input formats, and practical…", "heading": "4. Key Parameters", "content": "Page: Extracting JSON (LLM)\nSection: 4. Key Parameters\n\nBelow is an overview of important LLM extraction parameters. All are typically set inside `LLMExtractionStrategy(...)`. You then put that strategy in your `CrawlerRunConfig(...,…", "code_blocks": [{"language": "python", "code": "extraction_strategy = LLMExtractionStrategy(\n    llm_config = LLMConfig(provider=\"openai/gpt-4\", api_token=\"YOUR_OPENAI_KEY\"),\n    schema=MyModel.model_json_schema(),\n    extraction_type=\"schema\",…", "filename": ""}], "chunk_position": 57, "heading_path": "4. Key Parameters > 4. Key Parameters", "breadcrumbs": "Extracting JSON (LLM) > 4. Key Parameters > 4. Key Parameters"}, {"id": "8cbb9bc2651bac24", "url": "https://docs.crawl4ai.com/extraction/llm-strategies/", "page_title": "Extracting JSON (LLM)", "page_type": "guide", "page_summary": "A guide to using Crawl4AI's LLM-based extraction strategy to extract structured JSON from web pages using any LLM via LiteLLM, including schema definition, chunking, input formats, and practical…", "heading": "5. Putting It in `CrawlerRunConfig`", "content": "Page: Extracting JSON (LLM)\nSection: 5. Putting It in `CrawlerRunConfig`\n\n**Important**: In Crawl4AI, all strategy definitions should go inside the `CrawlerRunConfig`, not directly as a param in `arun()`. Here’s a full example:", "code_blocks": [{"language": "python", "code": "import os\nimport asyncio\nimport json\nfrom pydantic import BaseModel, Field\nfrom typing import List\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode, LLMConfig\nfrom…", "filename": ""}], "chunk_position": 57, "heading_path": "5. Putting It in `CrawlerRunConfig` > 5. Putting It in `CrawlerRunConfig`", "breadcrumbs": "Extracting JSON (LLM) > 5. Putting It in `CrawlerRunConfig` > 5. Putting It in `CrawlerRunConfig`"}, {"id": "1c6a95f1cc6d8261", "url": "https://docs.crawl4ai.com/extraction/llm-strategies/", "page_title": "Extracting JSON (LLM)", "page_type": "guide", "page_summary": "A guide to using Crawl4AI's LLM-based extraction strategy to extract structured JSON from web pages using any LLM via LiteLLM, including schema definition, chunking, input formats, and practical…", "heading": "6.1 `chunk_token_threshold`", "content": "Page: Extracting JSON (LLM)\nSection: 6.1 `chunk_token_threshold`\n\nIf your page is large, you might exceed your LLM’s context window. **`chunk_token_threshold`** sets the approximate max tokens per chunk. The library calculates word→token ratio using…", "code_blocks": [], "chunk_position": 57, "heading_path": "6.1 `chunk_token_threshold` > 6.1 `chunk_token_threshold`", "breadcrumbs": "Extracting JSON (LLM) > 6.1 `chunk_token_threshold` > 6.1 `chunk_token_threshold`"}, {"id": "c5ede7dad3fef56f", "url": "https://docs.crawl4ai.com/extraction/llm-strategies/", "page_title": "Extracting JSON (LLM)", "page_type": "guide", "page_summary": "A guide to using Crawl4AI's LLM-based extraction strategy to extract structured JSON from web pages using any LLM via LiteLLM, including schema definition, chunking, input formats, and practical…", "heading": "6.2 `overlap_rate`", "content": "Page: Extracting JSON (LLM)\nSection: 6.2 `overlap_rate`\n\nTo keep context continuous across chunks, we can overlap them. E.g., `overlap_rate=0.1` means each subsequent chunk includes 10% of the previous chunk’s text. This is helpful if your needed info…", "code_blocks": [], "chunk_position": 57, "heading_path": "6.2 `overlap_rate` > 6.2 `overlap_rate`", "breadcrumbs": "Extracting JSON (LLM) > 6.2 `overlap_rate` > 6.2 `overlap_rate`"}, {"id": "2a79a4a39155a179", "url": "https://docs.crawl4ai.com/extraction/llm-strategies/", "page_title": "Extracting JSON (LLM)", "page_type": "guide", "page_summary": "A guide to using Crawl4AI's LLM-based extraction strategy to extract structured JSON from web pages using any LLM via LiteLLM, including schema definition, chunking, input formats, and practical…", "heading": "6.3 Performance & Parallelism", "content": "Page: Extracting JSON (LLM)\nSection: 6.3 Performance & Parallelism\n\nBy chunking, you can potentially process multiple chunks in parallel (depending on your concurrency settings and the LLM provider). This reduces total time if the site is huge or has many sections.", "code_blocks": [], "chunk_position": 57, "heading_path": "6.3 Performance & Parallelism > 6.3 Performance & Parallelism", "breadcrumbs": "Extracting JSON (LLM) > 6.3 Performance & Parallelism > 6.3 Performance & Parallelism"}, {"id": "491307793c81465d", "url": "https://docs.crawl4ai.com/extraction/llm-strategies/", "page_title": "Extracting JSON (LLM)", "page_type": "guide", "page_summary": "A guide to using Crawl4AI's LLM-based extraction strategy to extract structured JSON from web pages using any LLM via LiteLLM, including schema definition, chunking, input formats, and practical…", "heading": "7. Input Format", "content": "Page: Extracting JSON (LLM)\nSection: 7. Input Format\n\nBy default, **LLMExtractionStrategy** uses `input_format=\"markdown\"`, meaning the **crawler’s final markdown** is fed to the LLM. You can change to:\n\n- **`html`**: The cleaned HTML or raw HTML…", "code_blocks": [{"language": "python", "code": "LLMExtractionStrategy(\n    # ...\n    input_format=\"html\",  # Instead of \"markdown\" or \"fit_markdown\"\n)", "filename": ""}], "chunk_position": 57, "heading_path": "7. Input Format > 7. Input Format", "breadcrumbs": "Extracting JSON (LLM) > 7. Input Format > 7. Input Format"}, {"id": "17721a0d3fd76563", "url": "https://docs.crawl4ai.com/extraction/llm-strategies/", "page_title": "Extracting JSON (LLM)", "page_type": "guide", "page_summary": "A guide to using Crawl4AI's LLM-based extraction strategy to extract structured JSON from web pages using any LLM via LiteLLM, including schema definition, chunking, input formats, and practical…", "heading": "8. Token Usage & Show Usage", "content": "Page: Extracting JSON (LLM)\nSection: 8. Token Usage & Show Usage\n\nTo keep track of tokens and cost, each chunk is processed with an LLM call. We record usage in:\n\n- **`usages`** (list): token usage per chunk or call.\n- **`total_usage`**: sum of all chunk calls.\n-…", "code_blocks": [{"language": "python", "code": "llm_strategy = LLMExtractionStrategy(...)\n# ...\nllm_strategy.show_usage()\n# e.g. “Total usage: 1241 tokens across 2 chunk calls”", "filename": ""}], "chunk_position": 57, "heading_path": "8. Token Usage & Show Usage > 8. Token Usage & Show Usage", "breadcrumbs": "Extracting JSON (LLM) > 8. Token Usage & Show Usage > 8. Token Usage & Show Usage"}, {"id": "f14f3526650d74f9", "url": "https://docs.crawl4ai.com/extraction/llm-strategies/", "page_title": "Extracting JSON (LLM)", "page_type": "guide", "page_summary": "A guide to using Crawl4AI's LLM-based extraction strategy to extract structured JSON from web pages using any LLM via LiteLLM, including schema definition, chunking, input formats, and practical…", "heading": "9. Example: Building a Knowledge Graph", "content": "Page: Extracting JSON (LLM)\nSection: 9. Example: Building a Knowledge Graph\n\nBelow is a snippet combining **`LLMExtractionStrategy`** with a Pydantic schema for a knowledge graph. Notice how we pass an **`instruction`** telling the model what to parse.\n\n**Key…", "code_blocks": [{"language": "python", "code": "import os\nimport json\nimport asyncio\nfrom typing import List\nfrom pydantic import BaseModel, Field\nfrom crawl4ai import AsyncWebCrawler, BrowserConfig, CrawlerRunConfig, CacheMode, LLMConfig\nfrom…", "filename": ""}], "chunk_position": 57, "heading_path": "9. Example: Building a Knowledge Graph > 9. Example: Building a Knowledge Graph", "breadcrumbs": "Extracting JSON (LLM) > 9. Example: Building a Knowledge Graph > 9. Example: Building a Knowledge Graph"}, {"id": "6a97ce9994b4a00b", "url": "https://docs.crawl4ai.com/extraction/llm-strategies/", "page_title": "Extracting JSON (LLM)", "page_type": "guide", "page_summary": "A guide to using Crawl4AI's LLM-based extraction strategy to extract structured JSON from web pages using any LLM via LiteLLM, including schema definition, chunking, input formats, and practical…", "heading": "10. Best Practices & Caveats", "content": "Page: Extracting JSON (LLM)\nSection: 10. Best Practices & Caveats\n\n1. **Cost & Latency**: LLM calls can be slow or expensive. Consider chunking or smaller coverage if you only need partial data.\n2. **Model Token Limits**: If your page + instruction exceed the…", "code_blocks": [], "chunk_position": 57, "heading_path": "10. Best Practices & Caveats > 10. Best Practices & Caveats", "breadcrumbs": "Extracting JSON (LLM) > 10. Best Practices & Caveats > 10. Best Practices & Caveats"}, {"id": "9cd25242b7052b34", "url": "https://docs.crawl4ai.com/extraction/llm-strategies/", "page_title": "Extracting JSON (LLM)", "page_type": "guide", "page_summary": "A guide to using Crawl4AI's LLM-based extraction strategy to extract structured JSON from web pages using any LLM via LiteLLM, including schema definition, chunking, input formats, and practical…", "heading": "11. Conclusion", "content": "Page: Extracting JSON (LLM)\nSection: 11. Conclusion\n\n**LLM-based extraction** in Crawl4AI is **provider-agnostic**, letting you choose from hundreds of models via LiteLLM. It’s perfect for **semantically complex** tasks or generating advanced…", "code_blocks": [], "chunk_position": 57, "heading_path": "11. Conclusion > 11. Conclusion", "breadcrumbs": "Extracting JSON (LLM) > 11. Conclusion > 11. Conclusion"}, {"id": "0e5afabc7ba8aedc", "url": "https://docs.crawl4ai.com/extraction/no-llm-strategies/", "page_title": "Extracting JSON (No LLM)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's LLM-free extraction strategies, including schema-based extraction with CSS/XPath selectors (JsonCssExtractionStrategy, JsonXPathExtractionStrategy) and regex-based…", "heading": "1. Intro to Schema-Based Extraction", "content": "Page: Extracting JSON (No LLM)\nSection: 1. Intro to Schema-Based Extraction\n\nA schema defines:\n\n- A **base selector** that identifies each \"container\" element on the page (e.g., a product row, a blog post card).\n- **Fields** describing which CSS/XPath selectors to use for…", "code_blocks": [], "chunk_position": 58, "heading_path": "1. Intro to Schema-Based Extraction > 1. Intro to Schema-Based Extraction", "breadcrumbs": "Extracting JSON (No LLM) > 1. Intro to Schema-Based Extraction > 1. Intro to Schema-Based Extraction"}, {"id": "6d64f30a9b05eb75", "url": "https://docs.crawl4ai.com/extraction/no-llm-strategies/", "page_title": "Extracting JSON (No LLM)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's LLM-free extraction strategies, including schema-based extraction with CSS/XPath selectors (JsonCssExtractionStrategy, JsonXPathExtractionStrategy) and regex-based…", "heading": "2. Simple Example: Crypto Prices", "content": "Page: Extracting JSON (No LLM)\nSection: 2. Simple Example: Crypto Prices\n\nLet's begin with a **simple** schema-based extraction using the `JsonCssExtractionStrategy`. Below is a snippet that extracts cryptocurrency prices from a site (similar to the legacy Coinbase…", "code_blocks": [{"language": "python", "code": "import json\nimport asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, CacheMode\nfrom crawl4ai import JsonCssExtractionStrategy\n\nasync def extract_crypto_prices():\n    # 1. Define a…", "filename": ""}, {"language": "python", "code": "import json\nimport asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai import JsonXPathExtractionStrategy\n\nasync def extract_crypto_prices_xpath():\n    # 1. Minimal dummy…", "filename": ""}], "chunk_position": 58, "heading_path": "2. Simple Example: Crypto Prices > 2. Simple Example: Crypto Prices", "breadcrumbs": "Extracting JSON (No LLM) > 2. Simple Example: Crypto Prices > 2. Simple Example: Crypto Prices"}, {"id": "51f4f960f87e13ad", "url": "https://docs.crawl4ai.com/extraction/no-llm-strategies/", "page_title": "Extracting JSON (No LLM)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's LLM-free extraction strategies, including schema-based extraction with CSS/XPath selectors (JsonCssExtractionStrategy, JsonXPathExtractionStrategy) and regex-based…", "heading": "3. Advanced Schema & Nested Structures", "content": "Page: Extracting JSON (No LLM)\nSection: 3. Advanced Schema & Nested Structures\n\nReal sites often have **nested** or repeated data—like categories containing products, which themselves have a list of reviews or features. For that, we can define **nested** or **list** (and even…", "code_blocks": [{"language": "python", "code": "schema = {\n    \"name\": \"E-commerce Product Catalog\",\n    \"baseSelector\": \"div.category\",\n    # (1) We can define optional baseFields if we want to extract attributes \n    # from the category…", "filename": ""}, {"language": "python", "code": "import json\nimport asyncio\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig\nfrom crawl4ai import JsonCssExtractionStrategy\n\necommerce_schema = {\n    # ... the advanced schema from above…", "filename": ""}], "chunk_position": 58, "heading_path": "3. Advanced Schema & Nested Structures > 3. Advanced Schema & Nested Structures", "breadcrumbs": "Extracting JSON (No LLM) > 3. Advanced Schema & Nested Structures > 3. Advanced Schema & Nested Structures"}, {"id": "c1ee925d2c1c7608", "url": "https://docs.crawl4ai.com/extraction/no-llm-strategies/", "page_title": "Extracting JSON (No LLM)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's LLM-free extraction strategies, including schema-based extraction with CSS/XPath selectors (JsonCssExtractionStrategy, JsonXPathExtractionStrategy) and regex-based…", "heading": "4. RegexExtractionStrategy - Fast Pattern-Based Extraction", "content": "Page: Extracting JSON (No LLM)\nSection: 4. RegexExtractionStrategy - Fast Pattern-Based Extraction\n\nCrawl4AI now offers a powerful new zero-LLM extraction strategy: `RegexExtractionStrategy`. This strategy provides lightning-fast extraction of common data types like emails, phone numbers, URLs,…", "code_blocks": [{"language": "python", "code": "import json\nimport asyncio\nfrom crawl4ai import (\n    AsyncWebCrawler,\n    CrawlerRunConfig,\n    RegexExtractionStrategy\n)\n\nasync def extract_with_regex():\n    # Create a strategy using built-in…", "filename": ""}, {"language": "python", "code": "# Use individual patterns\nstrategy = RegexExtractionStrategy(pattern=RegexExtractionStrategy.Email)\n\n# Combine multiple patterns\nstrategy = RegexExtractionStrategy(\n    pattern = (…", "filename": ""}, {"language": "python", "code": "import json\nimport asyncio\nfrom crawl4ai import (\n    AsyncWebCrawler,\n    CrawlerRunConfig,\n    RegexExtractionStrategy\n)\n\nasync def extract_prices():\n    # Define a custom pattern for US Dollar…", "filename": ""}, {"language": "python", "code": "import json\nimport asyncio\nfrom pathlib import Path\nfrom crawl4ai import (\n    AsyncWebCrawler,\n    CrawlerRunConfig,\n    RegexExtractionStrategy,\n    LLMConfig\n)\n\nasync def…", "filename": ""}, {"language": "json", "code": "[\n  {\n    \"url\": \"https://example.com\",\n    \"label\": \"email\",\n    \"value\": \"contact@example.com\",\n    \"span\": [145, 163]\n  },\n  {\n    \"url\": \"https://example.com\",\n    \"label\": \"url\",\n    \"value\":…", "filename": ""}], "chunk_position": 58, "heading_path": "4. RegexExtractionStrategy - Fast Pattern-Based Extraction > 4. RegexExtractionStrategy - Fast Pattern-Based Extraction", "breadcrumbs": "Extracting JSON (No LLM) > 4. RegexExtractionStrategy - Fast Pattern-Based Extraction > 4. RegexExtractionStrategy - Fast Pattern-Based Extraction"}, {"id": "0d98dd48687ccbec", "url": "https://docs.crawl4ai.com/extraction/no-llm-strategies/", "page_title": "Extracting JSON (No LLM)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's LLM-free extraction strategies, including schema-based extraction with CSS/XPath selectors (JsonCssExtractionStrategy, JsonXPathExtractionStrategy) and regex-based…", "heading": "5. Why \"No LLM\" Is Often Better", "content": "Page: Extracting JSON (No LLM)\nSection: 5. Why \"No LLM\" Is Often Better\n\n- **Zero Hallucination**: Pattern-based extraction doesn't guess text. It either finds it or not.\n- **Guaranteed Structure**: The same schema or regex yields consistent JSON across many pages, so…", "code_blocks": [], "chunk_position": 58, "heading_path": "5. Why \"No LLM\" Is Often Better > 5. Why \"No LLM\" Is Often Better", "breadcrumbs": "Extracting JSON (No LLM) > 5. Why \"No LLM\" Is Often Better > 5. Why \"No LLM\" Is Often Better"}, {"id": "41ae2b8fe5250a6d", "url": "https://docs.crawl4ai.com/extraction/no-llm-strategies/", "page_title": "Extracting JSON (No LLM)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's LLM-free extraction strategies, including schema-based extraction with CSS/XPath selectors (JsonCssExtractionStrategy, JsonXPathExtractionStrategy) and regex-based…", "heading": "6. Base Element Attributes & Additional Fields", "content": "Page: Extracting JSON (No LLM)\nSection: 6. Base Element Attributes & Additional Fields\n\nIt's easy to **extract attributes** (like `href`, `src`, or `data-xxx`) from your base or nested elements using:\n\n```json\n{\n  \"name\": \"href\",\n  \"type\": \"attribute\",\n  \"attribute\": \"href\",…", "code_blocks": [{"language": "json", "code": "{\n  \"name\": \"href\",\n  \"type\": \"attribute\",\n  \"attribute\": \"href\",\n  \"default\": null\n}", "filename": ""}], "chunk_position": 58, "heading_path": "6. Base Element Attributes & Additional Fields > 6. Base Element Attributes & Additional Fields", "breadcrumbs": "Extracting JSON (No LLM) > 6. Base Element Attributes & Additional Fields > 6. Base Element Attributes & Additional Fields"}, {"id": "5ec4fbfe5da553ac", "url": "https://docs.crawl4ai.com/extraction/no-llm-strategies/", "page_title": "Extracting JSON (No LLM)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's LLM-free extraction strategies, including schema-based extraction with CSS/XPath selectors (JsonCssExtractionStrategy, JsonXPathExtractionStrategy) and regex-based…", "heading": "7. Putting It All Together: Larger Example", "content": "Page: Extracting JSON (No LLM)\nSection: 7. Putting It All Together: Larger Example\n\nConsider a blog site. We have a schema that extracts the **URL** from each post card (via `baseFields` with an `\"attribute\": \"href\"`), plus the title, date, summary, and author:\n\n```json\nschema = {…", "code_blocks": [{"language": "json", "code": "schema = {\n  \"name\": \"Blog Posts\",\n  \"baseSelector\": \"a.blog-post-card\",\n  \"baseFields\": [\n    {\"name\": \"post_url\", \"type\": \"attribute\", \"attribute\": \"href\"}\n  ],\n  \"fields\": [\n    {\"name\": \"title\",…", "filename": ""}], "chunk_position": 58, "heading_path": "7. Putting It All Together: Larger Example > 7. Putting It All Together: Larger Example", "breadcrumbs": "Extracting JSON (No LLM) > 7. Putting It All Together: Larger Example > 7. Putting It All Together: Larger Example"}, {"id": "11d824748e1b456d", "url": "https://docs.crawl4ai.com/extraction/no-llm-strategies/", "page_title": "Extracting JSON (No LLM)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's LLM-free extraction strategies, including schema-based extraction with CSS/XPath selectors (JsonCssExtractionStrategy, JsonXPathExtractionStrategy) and regex-based…", "heading": "8. Extracting Sibling Data with `source`", "content": "Page: Extracting JSON (No LLM)\nSection: 8. Extracting Sibling Data with `source`\n\nSome websites split a single logical item across **sibling elements** rather than nesting everything inside one container. A classic example is Hacker News, where each submission spans two adjacent…", "code_blocks": [{"language": "html", "code": "<tr class=\"athing submission\">  <!-- rank, title, url -->\n  <td><span class=\"rank\">1.</span></td>\n  <td><span class=\"titleline\"><a href=\"https://example.com\">Example Title</a></span></td>\n</tr>\n<tr>…", "filename": ""}, {"language": "python", "code": "schema = {\n    \"name\": \"HN Submissions\",\n    \"baseSelector\": \"tr.athing.submission\",\n    \"fields\": [\n        {\"name\": \"rank\", \"selector\": \"span.rank\", \"type\": \"text\"},\n        {\"name\": \"title\",…", "filename": ""}], "chunk_position": 58, "heading_path": "8. Extracting Sibling Data with `source` > 8. Extracting Sibling Data with `source`", "breadcrumbs": "Extracting JSON (No LLM) > 8. Extracting Sibling Data with `source` > 8. Extracting Sibling Data with `source`"}, {"id": "b56324bc57653e49", "url": "https://docs.crawl4ai.com/extraction/no-llm-strategies/", "page_title": "Extracting JSON (No LLM)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's LLM-free extraction strategies, including schema-based extraction with CSS/XPath selectors (JsonCssExtractionStrategy, JsonXPathExtractionStrategy) and regex-based…", "heading": "9. Tips & Best Practices", "content": "Page: Extracting JSON (No LLM)\nSection: 9. Tips & Best Practices\n\n- **Inspect the DOM** in Chrome DevTools or Firefox's Inspector to find stable selectors.\n- **Start Simple**: Verify you can extract a single field. Then add complexity like nested objects or…", "code_blocks": [], "chunk_position": 58, "heading_path": "9. Tips & Best Practices > 9. Tips & Best Practices", "breadcrumbs": "Extracting JSON (No LLM) > 9. Tips & Best Practices > 9. Tips & Best Practices"}, {"id": "feb0786269c46994", "url": "https://docs.crawl4ai.com/extraction/no-llm-strategies/", "page_title": "Extracting JSON (No LLM)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's LLM-free extraction strategies, including schema-based extraction with CSS/XPath selectors (JsonCssExtractionStrategy, JsonXPathExtractionStrategy) and regex-based…", "heading": "10. Schema Generation Utility", "content": "Page: Extracting JSON (No LLM)\nSection: 10. Schema Generation Utility\n\nWhile manually crafting schemas is powerful and precise, Crawl4AI now offers a convenient utility to **automatically generate** extraction schemas using LLM. This is particularly useful when:\n\n-…", "code_blocks": [{"language": "python", "code": "from crawl4ai import JsonCssExtractionStrategy, JsonXPathExtractionStrategy\nfrom crawl4ai import LLMConfig\n\n# Sample HTML with product information\nhtml = \"\"\"\n<div class=\"product-card\">\n    <h2…", "filename": ""}, {"language": "python", "code": "# Default: validated (recommended)\nschema = JsonCssExtractionStrategy.generate_schema(\n    url=\"https://news.ycombinator.com\",\n    query=\"Extract each story: title, url, score, author\",\n)\n\n# Skip…", "filename": ""}, {"language": "python", "code": "from crawl4ai import JsonCssExtractionStrategy\nfrom crawl4ai.models import TokenUsage\n\nusage = TokenUsage()\n\nschema = JsonCssExtractionStrategy.generate_schema(…", "filename": ""}, {"language": "python", "code": "usage = TokenUsage()\nschema1 = JsonCssExtractionStrategy.generate_schema(url=url1, query=q1, usage=usage)\nschema2 = JsonCssExtractionStrategy.generate_schema(url=url2, query=q2,…", "filename": ""}, {"language": "python", "code": "from crawl4ai import JsonCssExtractionStrategy, LLMConfig\n\n# Collect HTML samples from different pages\nhtml_sample_1 = \"\"\"\n<table class=\"specs\">\n  <tr><td>Brand</td><td>Apple</td></tr>…", "filename": ""}], "chunk_position": 58, "heading_path": "10. Schema Generation Utility > 10. Schema Generation Utility", "breadcrumbs": "Extracting JSON (No LLM) > 10. Schema Generation Utility > 10. Schema Generation Utility"}, {"id": "73d3dac342a8c70d", "url": "https://docs.crawl4ai.com/extraction/no-llm-strategies/", "page_title": "Extracting JSON (No LLM)", "page_type": "guide", "page_summary": "This page covers Crawl4AI's LLM-free extraction strategies, including schema-based extraction with CSS/XPath selectors (JsonCssExtractionStrategy, JsonXPathExtractionStrategy) and regex-based…", "heading": "11. Conclusion", "content": "Page: Extracting JSON (No LLM)\nSection: 11. Conclusion\n\nWith Crawl4AI's LLM-free extraction strategies - `JsonCssExtractionStrategy`, `JsonXPathExtractionStrategy`, and now `RegexExtractionStrategy` - you can build powerful pipelines that:\n\n- Scrape any…", "code_blocks": [], "chunk_position": 58, "heading_path": "11. Conclusion > 11. Conclusion", "breadcrumbs": "Extracting JSON (No LLM) > 11. Conclusion > 11. Conclusion"}, {"id": "e0f15451295d7bd4", "url": "https://docs.crawl4ai.com/migration/table_extraction_v073/", "page_title": "Migration Guide: Table Extraction v0.7.3", "page_type": "guide", "page_summary": "A migration guide for Crawl4AI v0.7.3 introducing the Table Extraction Strategy Pattern, highlighting new classes and options, full backward compatibility, migration scenarios, code organization…", "heading": "Overview", "content": "Page: Migration Guide: Table Extraction v0.7.3\nSection: Overview\n\nVersion 0.7.3 introduces the **Table Extraction Strategy Pattern** , providing a more flexible and extensible approach to table extraction while maintaining full backward compatibility.", "code_blocks": [], "chunk_position": 59, "heading_path": "Overview > Overview", "breadcrumbs": "Migration Guide: Table Extraction v0.7.3 > Overview > Overview"}, {"id": "500cb2b21dfed1f3", "url": "https://docs.crawl4ai.com/migration/table_extraction_v073/", "page_title": "Migration Guide: Table Extraction v0.7.3", "page_type": "guide", "page_summary": "A migration guide for Crawl4AI v0.7.3 introducing the Table Extraction Strategy Pattern, highlighting new classes and options, full backward compatibility, migration scenarios, code organization…", "heading": "Strategy Pattern Implementation", "content": "Page: Migration Guide: Table Extraction v0.7.3\nSection: Strategy Pattern Implementation\n\nTable extraction now follows the same strategy pattern used throughout Crawl4AI:\n\n- **Consistent Architecture** : Aligns with extraction, chunking, and markdown strategies\n- **Extensibility** : Easy…", "code_blocks": [], "chunk_position": 59, "heading_path": "Strategy Pattern Implementation > Strategy Pattern Implementation", "breadcrumbs": "Migration Guide: Table Extraction v0.7.3 > Strategy Pattern Implementation > Strategy Pattern Implementation"}, {"id": "c9a199e852ee4616", "url": "https://docs.crawl4ai.com/migration/table_extraction_v073/", "page_title": "Migration Guide: Table Extraction v0.7.3", "page_type": "guide", "page_summary": "A migration guide for Crawl4AI v0.7.3 introducing the Table Extraction Strategy Pattern, highlighting new classes and options, full backward compatibility, migration scenarios, code organization…", "heading": "New Classes", "content": "Page: Migration Guide: Table Extraction v0.7.3\nSection: New Classes\n\n```python\nfrom crawl4ai import (\n    TableExtractionStrategy,    # Abstract base class\n    DefaultTableExtraction,      # Current implementation (default)\n    NoTableExtraction           # Explicitly disable…\n```", "code_blocks": [{"language": "python", "code": "from crawl4ai import (\n    TableExtractionStrategy,    # Abstract base class\n    DefaultTableExtraction,      # Current implementation (default)\n    NoTableExtraction           # Explicitly disable…", "filename": ""}], "chunk_position": 59, "heading_path": "New Classes > New Classes", "breadcrumbs": "Migration Guide: Table Extraction v0.7.3 > New Classes > New Classes"}, {"id": "406da21ac78c6527", "url": "https://docs.crawl4ai.com/migration/table_extraction_v073/", "page_title": "Migration Guide: Table Extraction v0.7.3", "page_type": "guide", "page_summary": "A migration guide for Crawl4AI v0.7.3 introducing the Table Extraction Strategy Pattern, highlighting new classes and options, full backward compatibility, migration scenarios, code organization…", "heading": "Backward Compatibility", "content": "Page: Migration Guide: Table Extraction v0.7.3\nSection: Backward Compatibility\n\n**✅ All existing code continues to work without changes.**", "code_blocks": [], "chunk_position": 59, "heading_path": "Backward Compatibility > Backward Compatibility", "breadcrumbs": "Migration Guide: Table Extraction v0.7.3 > Backward Compatibility > Backward Compatibility"}, {"id": "2fb520861fd32912", "url": "https://docs.crawl4ai.com/migration/table_extraction_v073/", "page_title": "Migration Guide: Table Extraction v0.7.3", "page_type": "guide", "page_summary": "A migration guide for Crawl4AI v0.7.3 introducing the Table Extraction Strategy Pattern, highlighting new classes and options, full backward compatibility, migration scenarios, code organization…", "heading": "No Changes Required", "content": "Page: Migration Guide: Table Extraction v0.7.3\nSection: No Changes Required\n\nIf your code looks like this, it will continue to work:", "code_blocks": [{"language": "python", "code": "# This still works exactly the same\nconfig = CrawlerRunConfig(\n    table_score_threshold=7\n)\nresult = await crawler.arun(url, config)\ntables = result.tables  # Same structure, same data", "filename": ""}], "chunk_position": 59, "heading_path": "No Changes Required > No Changes Required", "breadcrumbs": "Migration Guide: Table Extraction v0.7.3 > No Changes Required > No Changes Required"}, {"id": "d1024f47da61d8d8", "url": "https://docs.crawl4ai.com/migration/table_extraction_v073/", "page_title": "Migration Guide: Table Extraction v0.7.3", "page_type": "guide", "page_summary": "A migration guide for Crawl4AI v0.7.3 introducing the Table Extraction Strategy Pattern, highlighting new classes and options, full backward compatibility, migration scenarios, code organization…", "heading": "What Happens Behind the Scenes", "content": "Page: Migration Guide: Table Extraction v0.7.3\nSection: What Happens Behind the Scenes\n\nWhen you don't specify a `table_extraction` strategy:\n\n- `CrawlerRunConfig` automatically creates `DefaultTableExtraction`\n- It uses your `table_score_threshold` parameter\n- Tables are extracted…", "code_blocks": [], "chunk_position": 59, "heading_path": "What Happens Behind the Scenes > What Happens Behind the Scenes", "breadcrumbs": "Migration Guide: Table Extraction v0.7.3 > What Happens Behind the Scenes > What Happens Behind the Scenes"}, {"id": "5de7a831e4e2fdab", "url": "https://docs.crawl4ai.com/migration/table_extraction_v073/", "page_title": "Migration Guide: Table Extraction v0.7.3", "page_type": "guide", "page_summary": "A migration guide for Crawl4AI v0.7.3 introducing the Table Extraction Strategy Pattern, highlighting new classes and options, full backward compatibility, migration scenarios, code organization…", "heading": "1. Explicit Strategy Configuration", "content": "Page: Migration Guide: Table Extraction v0.7.3\nSection: 1. Explicit Strategy Configuration\n\nYou can now explicitly configure table extraction:", "code_blocks": [{"language": "python", "code": "# New: Explicit control\nstrategy = DefaultTableExtraction(\n    table_score_threshold=7,\n    min_rows=2,              # New: minimum row filter\n    min_cols=2,              # New: minimum column…", "filename": ""}], "chunk_position": 59, "heading_path": "1. Explicit Strategy Configuration > 1. Explicit Strategy Configuration", "breadcrumbs": "Migration Guide: Table Extraction v0.7.3 > 1. Explicit Strategy Configuration > 1. Explicit Strategy Configuration"}, {"id": "a64fe74ea3c10093", "url": "https://docs.crawl4ai.com/migration/table_extraction_v073/", "page_title": "Migration Guide: Table Extraction v0.7.3", "page_type": "guide", "page_summary": "A migration guide for Crawl4AI v0.7.3 introducing the Table Extraction Strategy Pattern, highlighting new classes and options, full backward compatibility, migration scenarios, code organization…", "heading": "No Performance Impact", "content": "Page: Migration Guide: Table Extraction v0.7.3\nSection: No Performance Impact\n\nFor existing code, performance remains identical:\n- Same extraction logic\n- Same scoring algorithm\n- Same processing time", "code_blocks": [], "chunk_position": 59, "heading_path": "No Performance Impact > No Performance Impact", "breadcrumbs": "Migration Guide: Table Extraction v0.7.3 > No Performance Impact > No Performance Impact"}, {"id": "3e3bda6ca5e44148", "url": "https://docs.crawl4ai.com/migration/table_extraction_v073/", "page_title": "Migration Guide: Table Extraction v0.7.3", "page_type": "guide", "page_summary": "A migration guide for Crawl4AI v0.7.3 introducing the Table Extraction Strategy Pattern, highlighting new classes and options, full backward compatibility, migration scenarios, code organization…", "heading": "No Deprecations", "content": "Page: Migration Guide: Table Extraction v0.7.3\nSection: No Deprecations\n\n- All existing parameters continue to work\n- `table_score_threshold` in `CrawlerRunConfig` is still supported\n- No breaking changes", "code_blocks": [], "chunk_position": 59, "heading_path": "No Deprecations > No Deprecations", "breadcrumbs": "Migration Guide: Table Extraction v0.7.3 > No Deprecations > No Deprecations"}, {"id": "5ce4fcb592b523a3", "url": "https://docs.crawl4ai.com/migration/table_extraction_v073/", "page_title": "Migration Guide: Table Extraction v0.7.3", "page_type": "guide", "page_summary": "A migration guide for Crawl4AI v0.7.3 introducing the Table Extraction Strategy Pattern, highlighting new classes and options, full backward compatibility, migration scenarios, code organization…", "heading": "Internal Changes (Transparent to Users)", "content": "Page: Migration Guide: Table Extraction v0.7.3\nSection: Internal Changes (Transparent to Users)\n\n- `LXMLWebScrapingStrategy.is_data_table()` - Moved to `DefaultTableExtraction`\n- `LXMLWebScrapingStrategy.extract_table_data()` - Moved to `DefaultTableExtraction`\n\nThese methods were internal and…", "code_blocks": [], "chunk_position": 59, "heading_path": "Internal Changes (Transparent to Users) > Internal Changes (Transparent to Users)", "breadcrumbs": "Migration Guide: Table Extraction v0.7.3 > Internal Changes (Transparent to Users) > Internal Changes (Transparent to Users)"}, {"id": "767b2c3a2f4874f8", "url": "https://docs.crawl4ai.com/migration/table_extraction_v073/", "page_title": "Migration Guide: Table Extraction v0.7.3", "page_type": "guide", "page_summary": "A migration guide for Crawl4AI v0.7.3 introducing the Table Extraction Strategy Pattern, highlighting new classes and options, full backward compatibility, migration scenarios, code organization…", "heading": "Benefits of Upgrading", "content": "Page: Migration Guide: Table Extraction v0.7.3\nSection: Benefits of Upgrading\n\nWhile not required, using the new pattern provides:\n\n- **Better Control** : Filter tables during extraction, not after\n- **Performance Options** : Skip extraction when not needed\n- **Extensibility**…", "code_blocks": [], "chunk_position": 59, "heading_path": "Benefits of Upgrading > Benefits of Upgrading", "breadcrumbs": "Migration Guide: Table Extraction v0.7.3 > Benefits of Upgrading > Benefits of Upgrading"}, {"id": "3f5f4c5cb88a9ee8", "url": "https://docs.crawl4ai.com/migration/table_extraction_v073/", "page_title": "Migration Guide: Table Extraction v0.7.3", "page_type": "guide", "page_summary": "A migration guide for Crawl4AI v0.7.3 introducing the Table Extraction Strategy Pattern, highlighting new classes and options, full backward compatibility, migration scenarios, code organization…", "heading": "Issue: Different Number of Tables", "content": "Page: Migration Guide: Table Extraction v0.7.3\nSection: Issue: Different Number of Tables\n\n**Cause** : Threshold or filtering differences\n\n**Solution** :", "code_blocks": [{"language": "python", "code": "# Ensure same threshold\nstrategy = DefaultTableExtraction(\n    table_score_threshold=7,  # Match your old setting\n    min_rows=0,               # No filtering (default)\n    min_cols=0…", "filename": ""}], "chunk_position": 59, "heading_path": "Issue: Different Number of Tables > Issue: Different Number of Tables", "breadcrumbs": "Migration Guide: Table Extraction v0.7.3 > Issue: Different Number of Tables > Issue: Different Number of Tables"}, {"id": "f652b85aadca70f5", "url": "https://docs.crawl4ai.com/migration/table_extraction_v073/", "page_title": "Migration Guide: Table Extraction v0.7.3", "page_type": "guide", "page_summary": "A migration guide for Crawl4AI v0.7.3 introducing the Table Extraction Strategy Pattern, highlighting new classes and options, full backward compatibility, migration scenarios, code organization…", "heading": "Issue: Import Errors", "content": "Page: Migration Guide: Table Extraction v0.7.3\nSection: Issue: Import Errors\n\n**Cause** : Using new classes without importing\n\n**Solution** :", "code_blocks": [{"language": "python", "code": "# Add imports if using new features\nfrom crawl4ai import (\n    DefaultTableExtraction,\n    NoTableExtraction,\n    TableExtractionStrategy\n)", "filename": ""}], "chunk_position": 59, "heading_path": "Issue: Import Errors > Issue: Import Errors", "breadcrumbs": "Migration Guide: Table Extraction v0.7.3 > Issue: Import Errors > Issue: Import Errors"}, {"id": "3099373f3249363d", "url": "https://docs.crawl4ai.com/migration/table_extraction_v073/", "page_title": "Migration Guide: Table Extraction v0.7.3", "page_type": "guide", "page_summary": "A migration guide for Crawl4AI v0.7.3 introducing the Table Extraction Strategy Pattern, highlighting new classes and options, full backward compatibility, migration scenarios, code organization…", "heading": "Issue: Custom Strategy Not Working", "content": "Page: Migration Guide: Table Extraction v0.7.3\nSection: Issue: Custom Strategy Not Working\n\n**Cause** : Incorrect method signature\n\n**Solution** :", "code_blocks": [{"language": "python", "code": "class CustomExtractor(TableExtractionStrategy):\n    def extract_tables(self, element, **kwargs):  # Correct signature\n        # Not: extract_tables(self, html)\n        # Not: extract(self, element)…", "filename": ""}], "chunk_position": 59, "heading_path": "Issue: Custom Strategy Not Working > Issue: Custom Strategy Not Working", "breadcrumbs": "Migration Guide: Table Extraction v0.7.3 > Issue: Custom Strategy Not Working > Issue: Custom Strategy Not Working"}, {"id": "bc85f00603b503fa", "url": "https://docs.crawl4ai.com/migration/table_extraction_v073/", "page_title": "Migration Guide: Table Extraction v0.7.3", "page_type": "guide", "page_summary": "A migration guide for Crawl4AI v0.7.3 introducing the Table Extraction Strategy Pattern, highlighting new classes and options, full backward compatibility, migration scenarios, code organization…", "heading": "Getting Help", "content": "Page: Migration Guide: Table Extraction v0.7.3\nSection: Getting Help\n\nIf you encounter issues:\n\n- Check your `table_score_threshold` matches previous settings\n- Verify imports if using new classes\n- Enable verbose logging: `DefaultTableExtraction(verbose=True)`\n-…", "code_blocks": [], "chunk_position": 59, "heading_path": "Getting Help > Getting Help", "breadcrumbs": "Migration Guide: Table Extraction v0.7.3 > Getting Help > Getting Help"}, {"id": "3f9283a1719a0124", "url": "https://docs.crawl4ai.com/migration/table_extraction_v073/", "page_title": "Migration Guide: Table Extraction v0.7.3", "page_type": "guide", "page_summary": "A migration guide for Crawl4AI v0.7.3 introducing the Table Extraction Strategy Pattern, highlighting new classes and options, full backward compatibility, migration scenarios, code organization…", "heading": "Summary", "content": "Page: Migration Guide: Table Extraction v0.7.3\nSection: Summary\n\n- ✅  **Full backward compatibility**  - No code changes required\n- ✅  **Same results**  - Identical extraction behavior by default\n- ✅  **New options**  - Additional control when needed\n- ✅  **Better…", "code_blocks": [], "chunk_position": 59, "heading_path": "Summary > Summary", "breadcrumbs": "Migration Guide: Table Extraction v0.7.3 > Summary > Summary"}, {"id": "63bf9adcca7ac3a6", "url": "https://docs.crawl4ai.com/migration/webscraping-strategy-migration/", "page_title": "WebScrapingStrategy Migration Guide", "page_type": "guide", "page_summary": "This guide explains the deprecation of BeautifulSoup-based WebScrapingStrategy in favor of LXMLWebScrapingStrategy, and confirms backward compatibility with no required changes.", "heading": "Overview", "content": "Page: WebScrapingStrategy Migration Guide\nSection: Overview\n\nCrawl4AI has simplified its content scraping architecture. The BeautifulSoup-based `WebScrapingStrategy` has been deprecated in favor of the faster LXML-based implementation. However, **no action is…", "code_blocks": [], "chunk_position": 60, "heading_path": "Overview > Overview", "breadcrumbs": "WebScrapingStrategy Migration Guide > Overview > Overview"}, {"id": "41c6c1a93b6301fc", "url": "https://docs.crawl4ai.com/migration/webscraping-strategy-migration/", "page_title": "WebScrapingStrategy Migration Guide", "page_type": "guide", "page_summary": "This guide explains the deprecation of BeautifulSoup-based WebScrapingStrategy in favor of LXMLWebScrapingStrategy, and confirms backward compatibility with no required changes.", "heading": "What Changed?", "content": "Page: WebScrapingStrategy Migration Guide\nSection: What Changed?\n\n- **`WebScrapingStrategy` is now an alias** for `LXMLWebScrapingStrategy`\n- **The BeautifulSoup implementation has been removed** (~1000 lines of redundant code)\n- **`LXMLWebScrapingStrategy`…", "code_blocks": [], "chunk_position": 60, "heading_path": "What Changed? > What Changed?", "breadcrumbs": "WebScrapingStrategy Migration Guide > What Changed? > What Changed?"}, {"id": "6e5a7bb5ea521b21", "url": "https://docs.crawl4ai.com/migration/webscraping-strategy-migration/", "page_title": "WebScrapingStrategy Migration Guide", "page_type": "guide", "page_summary": "This guide explains the deprecation of BeautifulSoup-based WebScrapingStrategy in favor of LXMLWebScrapingStrategy, and confirms backward compatibility with no required changes.", "heading": "Backward Compatibility", "content": "Page: WebScrapingStrategy Migration Guide\nSection: Backward Compatibility\n\n**Your existing code continues to work without any changes:**", "code_blocks": [{"language": "python", "code": "# This still works perfectly\nfrom crawl4ai import AsyncWebCrawler, CrawlerRunConfig, WebScrapingStrategy\n\nconfig = CrawlerRunConfig(\n    scraping_strategy=WebScrapingStrategy()  # Works as before\n)", "filename": ""}], "chunk_position": 60, "heading_path": "Backward Compatibility > Backward Compatibility", "breadcrumbs": "WebScrapingStrategy Migration Guide > Backward Compatibility > Backward Compatibility"}, {"id": "c8f8720ea590f650", "url": "https://docs.crawl4ai.com/migration/webscraping-strategy-migration/", "page_title": "WebScrapingStrategy Migration Guide", "page_type": "guide", "page_summary": "This guide explains the deprecation of BeautifulSoup-based WebScrapingStrategy in favor of LXMLWebScrapingStrategy, and confirms backward compatibility with no required changes.", "heading": "Option 1: Do Nothing (Recommended)", "content": "Page: WebScrapingStrategy Migration Guide\nSection: Option 1: Do Nothing (Recommended)\n\nYour code will continue to work. `WebScrapingStrategy` is permanently aliased to `LXMLWebScrapingStrategy`.", "code_blocks": [], "chunk_position": 60, "heading_path": "Option 1: Do Nothing (Recommended) > Option 1: Do Nothing (Recommended)", "breadcrumbs": "WebScrapingStrategy Migration Guide > Option 1: Do Nothing (Recommended) > Option 1: Do Nothing (Recommended)"}, {"id": "fa4c45c1e9bfa3f4", "url": "https://docs.crawl4ai.com/migration/webscraping-strategy-migration/", "page_title": "WebScrapingStrategy Migration Guide", "page_type": "guide", "page_summary": "This guide explains the deprecation of BeautifulSoup-based WebScrapingStrategy in favor of LXMLWebScrapingStrategy, and confirms backward compatibility with no required changes.", "heading": "Option 3: Use Default Configuration", "content": "Page: WebScrapingStrategy Migration Guide\nSection: Option 3: Use Default Configuration\n\nSince `LXMLWebScrapingStrategy` is the default, you can omit the strategy parameter:", "code_blocks": [{"language": "python", "code": "# Simplest approach - uses LXMLWebScrapingStrategy by default\nconfig = CrawlerRunConfig()", "filename": ""}], "chunk_position": 60, "heading_path": "Option 3: Use Default Configuration > Option 3: Use Default Configuration", "breadcrumbs": "WebScrapingStrategy Migration Guide > Option 3: Use Default Configuration > Option 3: Use Default Configuration"}, {"id": "36c7e58e2294728e", "url": "https://docs.crawl4ai.com/migration/webscraping-strategy-migration/", "page_title": "WebScrapingStrategy Migration Guide", "page_type": "guide", "page_summary": "This guide explains the deprecation of BeautifulSoup-based WebScrapingStrategy in favor of LXMLWebScrapingStrategy, and confirms backward compatibility with no required changes.", "heading": "Subclassing", "content": "Page: WebScrapingStrategy Migration Guide\nSection: Subclassing\n\nIf you've subclassed `WebScrapingStrategy`, it continues to work:", "code_blocks": [{"language": "python", "code": "class MyCustomStrategy(WebScrapingStrategy):\n    def __init__(self):\n        super().__init__()\n        # Your custom code", "filename": ""}], "chunk_position": 60, "heading_path": "Subclassing > Subclassing", "breadcrumbs": "WebScrapingStrategy Migration Guide > Subclassing > Subclassing"}, {"id": "0aa348b4002902fd", "url": "https://docs.crawl4ai.com/migration/webscraping-strategy-migration/", "page_title": "WebScrapingStrategy Migration Guide", "page_type": "guide", "page_summary": "This guide explains the deprecation of BeautifulSoup-based WebScrapingStrategy in favor of LXMLWebScrapingStrategy, and confirms backward compatibility with no required changes.", "heading": "Performance Benefits", "content": "Page: WebScrapingStrategy Migration Guide\nSection: Performance Benefits\n\n- **10-20x faster** HTML parsing for large documents\n- **Lower memory usage**\n- **Consistent behavior** across all use cases\n- **Simplified maintenance** and bug fixes", "code_blocks": [], "chunk_position": 60, "heading_path": "Performance Benefits > Performance Benefits", "breadcrumbs": "WebScrapingStrategy Migration Guide > Performance Benefits > Performance Benefits"}, {"id": "1579a79174a158c9", "url": "https://docs.crawl4ai.com/migration/webscraping-strategy-migration/", "page_title": "WebScrapingStrategy Migration Guide", "page_type": "guide", "page_summary": "This guide explains the deprecation of BeautifulSoup-based WebScrapingStrategy in favor of LXMLWebScrapingStrategy, and confirms backward compatibility with no required changes.", "heading": "Summary", "content": "Page: WebScrapingStrategy Migration Guide\nSection: Summary\n\nThis change simplifies Crawl4AI's internals while maintaining 100% backward compatibility. Your existing code continues to work, and you get better performance automatically.", "code_blocks": [], "chunk_position": 60, "heading_path": "Summary > Summary", "breadcrumbs": "WebScrapingStrategy Migration Guide > Summary > Summary"}]