How to Crawl a Website in Python | UnWeb

Scraping a list of known URLs and crawling a website are different problems. Scraping is targeted: you know what you want. Crawling is discovery-driven: you start with a seed URL and follow links to find everything the site contains.

This guide builds a complete, production-ready web crawler in Python from first principles: BFS traversal, link extraction, URL normalization, deduplication, scope control, and async concurrency — all assembled into a reusable WebCrawler class.

The Core Crawling Algorithm

A web crawler is essentially breadth-first search over URLs:

  1. Add the seed URL to a queue
  2. Pop a URL from the queue
  3. Fetch the page
  4. Extract all links from the page
  5. Filter links to those within scope (same domain, matching patterns)
  6. Add unseen links to the queue
  7. Repeat from step 2 until the queue is empty or the limit is reached

The key challenges are: link extraction and normalization (step 4), deduplication (step 6), scope control, and doing all this efficiently with async concurrency.

Before building the crawler, we need reliable link extraction. Links come in many forms — absolute, relative, with fragments, with query strings — and need to be normalized before deduplication works correctly:

import re
from urllib.parse import urljoin, urlparse, urlunparse

def extract_links(base_url: str, html: str) -> set[str]:
    """Extract and normalize all links from an HTML page."""
    links = set()
    # Match href attributes (handles single and double quotes)
    pattern = re.compile(r'href=["\']([^"\']+)["\']', re.IGNORECASE)

    for match in pattern.finditer(html):
        raw_href = match.group(1).strip()

        # Skip non-HTTP links
        if raw_href.startswith(('mailto:', 'tel:', 'javascript:', '#')):
            continue

        # Resolve relative URLs against the base
        absolute = urljoin(base_url, raw_href)
        normalized = normalize_url(absolute)
        if normalized:
            links.add(normalized)

    return links


def normalize_url(url: str) -> str | None:
    """Normalize a URL for consistent deduplication."""
    try:
        parsed = urlparse(url)

        # Only crawl HTTP/HTTPS
        if parsed.scheme not in ('http', 'https'):
            return None

        # Remove fragment — same page with different anchor is the same page
        normalized = parsed._replace(fragment='')

        # Lowercase the scheme and host
        normalized = normalized._replace(
            scheme=normalized.scheme.lower(),
            netloc=normalized.netloc.lower(),
        )

        return urlunparse(normalized)
    except Exception:
        return None

Scope Control: Staying Within the Target Site

Without scope control, a crawler will follow links to external sites and never stop. At minimum, restrict crawling to the same domain:

from urllib.parse import urlparse

def is_in_scope(url: str, seed_url: str, allow_subdomains: bool = False) -> bool:
    """Check if a URL is within the crawl scope."""
    seed_domain = urlparse(seed_url).netloc
    url_domain = urlparse(url).netloc

    if allow_subdomains:
        # Allow example.com and blog.example.com
        base = seed_domain.lstrip('www.')
        return url_domain.endswith(base)
    else:
        # Strict: same domain only (normalize www)
        return url_domain.lstrip('www.') == seed_domain.lstrip('www.')


def matches_include_patterns(url: str, patterns: list[str]) -> bool:
    """Optionally restrict crawling to URLs matching specific path patterns."""
    if not patterns:
        return True
    return any(re.search(pattern, url) for pattern in patterns)


def matches_exclude_patterns(url: str, patterns: list[str]) -> bool:
    """Exclude URLs matching specific patterns (e.g., login, logout, admin)."""
    return any(re.search(pattern, url) for pattern in patterns)

The Async Web Crawler

Now we assemble the crawler. Using asyncio gives us concurrent fetching without the overhead of multiple threads:

import asyncio
import time
import random
from collections import deque
from dataclasses import dataclass, field
from urllib.parse import urlparse

import httpx

@dataclass
class CrawlResult:
    url: str
    status_code: int
    content: str
    content_type: str
    links_found: set[str]
    crawl_time: float
    error: str | None = None


class WebCrawler:
    def __init__(
        self,
        seed_url: str,
        max_pages: int = 100,
        max_concurrency: int = 3,
        delay_range: tuple[float, float] = (1.0, 2.5),
        include_patterns: list[str] | None = None,
        exclude_patterns: list[str] | None = None,
        allow_subdomains: bool = False,
        user_agent: str = "PythonCrawler/1.0",
    ):
        self.seed_url = seed_url
        self.max_pages = max_pages
        self.max_concurrency = max_concurrency
        self.delay_range = delay_range
        self.include_patterns = include_patterns or []
        self.exclude_patterns = exclude_patterns or [
            r'/logout', r'/login', r'/cart', r'/checkout',
            r'\.(pdf|zip|exe|dmg|pkg|deb|rpm)$',
        ]
        self.allow_subdomains = allow_subdomains
        self.user_agent = user_agent

        self._queue: deque[str] = deque([seed_url])
        self._seen: set[str] = {normalize_url(seed_url)}
        self._results: list[CrawlResult] = []

    async def crawl(self) -> list[CrawlResult]:
        semaphore = asyncio.Semaphore(self.max_concurrency)

        async with httpx.AsyncClient(
            headers={
                "User-Agent": self.user_agent,
                "Accept": "text/html,application/xhtml+xml;q=0.9,*/*;q=0.8",
            },
            follow_redirects=True,
            timeout=30.0,
        ) as client:
            while self._queue and len(self._results) < self.max_pages:
                # Collect a batch of URLs to process concurrently
                batch = []
                while self._queue and len(batch) < self.max_concurrency:
                    batch.append(self._queue.popleft())

                tasks = [self._crawl_url(client, url, semaphore) for url in batch]
                batch_results = await asyncio.gather(*tasks, return_exceptions=True)

                for result in batch_results:
                    if isinstance(result, CrawlResult):
                        self._results.append(result)
                        # Enqueue newly discovered links
                        for link in result.links_found:
                            if link not in self._seen:
                                self._seen.add(link)
                                self._queue.append(link)

        return self._results

    async def _crawl_url(
        self,
        client: httpx.AsyncClient,
        url: str,
        semaphore: asyncio.Semaphore,
    ) -> CrawlResult:
        async with semaphore:
            start = time.monotonic()
            try:
                response = await client.get(url)
                elapsed = time.monotonic() - start

                content = response.text
                content_type = response.headers.get("content-type", "")

                # Only extract links from HTML pages
                links = set()
                if "text/html" in content_type:
                    raw_links = extract_links(url, content)
                    links = {
                        link for link in raw_links
                        if is_in_scope(link, self.seed_url, self.allow_subdomains)
                        and not matches_exclude_patterns(link, self.exclude_patterns)
                        and matches_include_patterns(link, self.include_patterns)
                    }

                # Polite delay after each request
                delay = random.uniform(*self.delay_range)
                await asyncio.sleep(delay)

                return CrawlResult(
                    url=url,
                    status_code=response.status_code,
                    content=content,
                    content_type=content_type,
                    links_found=links,
                    crawl_time=elapsed,
                )

            except Exception as e:
                return CrawlResult(
                    url=url, status_code=0, content="", content_type="",
                    links_found=set(), crawl_time=time.monotonic() - start,
                    error=str(e),
                )

Using the Crawler

import asyncio

async def main():
    crawler = WebCrawler(
        seed_url="https://example.com",
        max_pages=200,
        max_concurrency=3,
        delay_range=(1.0, 2.5),
        exclude_patterns=[r'/logout', r'/tag/', r'/author/'],
    )

    results = await crawler.crawl()

    successful = [r for r in results if r.status_code == 200]
    failed = [r for r in results if r.error or r.status_code >= 400]

    print(f"Crawled: {len(results)} pages")
    print(f"Successful: {len(successful)}")
    print(f"Failed: {len(failed)}")
    print(f"Average crawl time: {sum(r.crawl_time for r in successful) / len(successful):.2f}s")

asyncio.run(main())

Persisting Results to SQLite

For crawls over a few dozen pages, you want results persisted to disk so you can resume after interruption and query results without re-crawling:

import sqlite3
from datetime import datetime

def init_db(db_path: str) -> sqlite3.Connection:
    conn = sqlite3.connect(db_path)
    conn.execute("""
        CREATE TABLE IF NOT EXISTS crawl_results (
            url TEXT PRIMARY KEY,
            status_code INTEGER,
            content TEXT,
            content_type TEXT,
            crawled_at TEXT,
            crawl_time_ms INTEGER,
            error TEXT
        )
    """)
    conn.commit()
    return conn

def save_result(conn: sqlite3.Connection, result: CrawlResult) -> None:
    conn.execute(
        """
        INSERT OR REPLACE INTO crawl_results
        (url, status_code, content, content_type, crawled_at, crawl_time_ms, error)
        VALUES (?, ?, ?, ?, ?, ?, ?)
        """,
        (
            result.url,
            result.status_code,
            result.content,
            result.content_type,
            datetime.utcnow().isoformat(),
            int(result.crawl_time * 1000),
            result.error,
        )
    )
    conn.commit()

def get_already_crawled(conn: sqlite3.Connection) -> set[str]:
    rows = conn.execute("SELECT url FROM crawl_results").fetchall()
    return {row[0] for row in rows}

To make the crawler resumable, seed the _seen set with already-crawled URLs from SQLite before starting.

Using UnWeb for Clean Content Extraction

The crawler above gives you raw HTML. For applications that need clean content (RAG pipelines, content analysis, search indexes), converting each page to Markdown via UnWeb is a natural integration:

UNWEB_API_KEY = "your_api_key"

async def fetch_as_markdown(client: httpx.AsyncClient, url: str) -> str:
    response = await client.get(
        "https://api.unweb.info/v1/convert",
        params={"url": url},
        headers={"Authorization": f"Bearer {UNWEB_API_KEY}"},
        timeout=30.0,
    )
    response.raise_for_status()
    return response.json()["markdown"]

# In the crawler, replace client.get(url) with fetch_as_markdown
# to get clean, structured content instead of raw HTML

This is particularly valuable for documentation crawlers, knowledge base builders, and any pipeline that feeds into an LLM — Markdown is dramatically cleaner input than raw HTML for these use cases.

Turn any crawled page into clean Markdown

UnWeb converts JS-rendered pages, complex layouts, and any URL to clean Markdown — ready for RAG, search indexes, or analysis. Free tier available.

Get your API key

Crawl Quality Checklist