diff --git a/deploy/docker/api.py b/deploy/docker/api.py index 7c90fb5ba..20ab6d508 100644 --- a/deploy/docker/api.py +++ b/deploy/docker/api.py @@ -647,6 +647,57 @@ async def stream_results(crawler: AsyncWebCrawler, results_gen: AsyncGenerator) await _dispose_crawler(crawler) +def _load_crawler_configs( + crawler_configs: List[dict], + base_config: Optional[dict] = None, + stream: bool = False, +) -> List[CrawlerRunConfig]: + """Deserialize a per-URL crawler_configs list under the untrusted gate. + Shared by the streaming and non-streaming crawl handlers so both apply the + same boundary and wire the PDF URL validator into every entry.""" + from crawl4ai.processors.pdf import PDFContentScrapingStrategy + + config_list = [ + CrawlerRunConfig.load(cc, provenance=Provenance.UNTRUSTED) + for cc in crawler_configs + ] + for cfg in config_list: + for key, value in (base_config or {}).items(): + if hasattr(cfg, key): + current_value = getattr(cfg, key) + if current_value is None or current_value == "": + setattr(cfg, key, value) + # SSRF: per-URL PDF strategies need the validator wired too + if isinstance(cfg.scraping_strategy, PDFContentScrapingStrategy): + cfg.scraping_strategy.url_validator = validate_url_destination + if stream: + # arun_many takes the stream flag from the first config of a list. + cfg.stream = True + return config_list + + +def _needs_pdf_crawler( + crawler_config: CrawlerRunConfig, + config_list: Optional[List[CrawlerRunConfig]] = None, +) -> bool: + """Whether the crawl runs on PDFCrawlerStrategy. A per-URL list decides when + present, since it is what arun_many runs. One crawler serves every URL of a + request, so a list mixing PDF and browser entries is refused rather than run + on the wrong one.""" + from crawl4ai.processors.pdf import PDFContentScrapingStrategy + + configs = config_list or [crawler_config] + pdf = [ + isinstance(cfg.scraping_strategy, PDFContentScrapingStrategy) for cfg in configs + ] + if any(pdf) and not all(pdf): + raise HTTPException( + status_code=400, + detail="crawler_configs cannot mix PDFContentScrapingStrategy with other scraping strategies in one request", + ) + return all(pdf) + + def _normalize_and_validate_seeds(urls: List[str]) -> List[str]: """Prefix bare hosts with https:// and SSRF-validate every seed URL's destination. Shared by the streaming and non-streaming crawl handlers so a @@ -699,42 +750,39 @@ async def handle_crawl_request( ) if config["crawler"]["rate_limiter"]["enabled"] else None ) + base_config = config["crawler"]["base_config"] + # Per-URL config list: deserialize each and apply base_config. Loaded + # before a crawler is chosen, because the list decides which one runs. + config_list = ( + _load_crawler_configs(crawler_configs, base_config) + if crawler_configs + else None + ) + from crawler_pool import get_crawler, release_crawler from crawl4ai.processors.pdf import PDFContentScrapingStrategy, PDFCrawlerStrategy - is_pdf_crawl = isinstance(crawler_config.scraping_strategy, PDFContentScrapingStrategy) + is_pdf_crawl = _needs_pdf_crawler(crawler_config, config_list) if is_pdf_crawl: if hooks_config: # PDFCrawlerStrategy has no browser page, so hooks can't attach to it raise HTTPException(status_code=400, detail="Hooks are not supported with PDFContentScrapingStrategy") # SSRF protection: vet the PDF download URL and every redirect hop - crawler_config.scraping_strategy.url_validator = validate_url_destination + if isinstance(crawler_config.scraping_strategy, PDFContentScrapingStrategy): + crawler_config.scraping_strategy.url_validator = validate_url_destination # Use PDFCrawlerStrategy when scraping PDFs, as headless Chromium can't render PDFs inline crawler = AsyncWebCrawler(crawler_strategy=PDFCrawlerStrategy()) await crawler.start() else: crawler = await get_crawler(browser_config) - + # Attach declarative hooks if provided hooks_status = {} if hooks_config: hooks_status = _attach_declarative_hooks(crawler, hooks_config) logger.info(f"Hooks attachment status: {hooks_status['status']}") - - base_config = config["crawler"]["base_config"] # Build the config(s) to pass to arun/arun_many - if crawler_configs and len(urls) > 1: - # Per-URL config list: deserialize each and apply base_config - config_list = [CrawlerRunConfig.load(cc, provenance=Provenance.UNTRUSTED) for cc in crawler_configs] - for cfg in config_list: - for key, value in base_config.items(): - if hasattr(cfg, key): - current_value = getattr(cfg, key) - if current_value is None or current_value == "": - setattr(cfg, key, value) - # SSRF: per-URL PDF strategies need the validator wired too - if isinstance(cfg.scraping_strategy, PDFContentScrapingStrategy): - cfg.scraping_strategy.url_validator = validate_url_destination + if config_list: effective_config = config_list else: # Single config (original behavior) @@ -746,9 +794,12 @@ async def handle_crawl_request( effective_config = crawler_config results = [] - func = getattr(crawler, "arun" if len(urls) == 1 else "arun_many") + # arun takes one config, so a per-URL list goes to arun_many even for + # a single URL; arun_many matches each URL against the list. + use_many = len(urls) > 1 or isinstance(effective_config, list) + func = getattr(crawler, "arun_many" if use_many else "arun") partial_func = partial(func, - urls[0] if len(urls) == 1 else urls, + urls if use_many else urls[0], config=effective_config, dispatcher=dispatcher) # Optional per-crawl wall-clock deadline (config limits.wall_clock_s; 0 = none). @@ -891,7 +942,8 @@ async def handle_stream_crawl_request( browser_config: dict, crawler_config: dict, config: dict, - hooks_config: Optional[dict] = None + hooks_config: Optional[dict] = None, + crawler_configs: Optional[List[dict]] = None, ) -> Tuple[AsyncWebCrawler, AsyncGenerator, Optional[Dict]]: """Handle streaming crawl requests with optional hooks.""" hooks_info = None @@ -916,6 +968,11 @@ async def handle_stream_crawl_request( clamp_deep_crawl(crawler_config) crawler_config.stream = True + config_list = ( + _load_crawler_configs(crawler_configs, stream=True) + if crawler_configs + else None + ) # Deep crawl streaming supports exactly one start URL if crawler_config.deep_crawl_strategy is not None and len(urls) != 1: @@ -929,12 +986,13 @@ async def handle_stream_crawl_request( from crawler_pool import get_crawler from crawl4ai.processors.pdf import PDFContentScrapingStrategy, PDFCrawlerStrategy - if isinstance(crawler_config.scraping_strategy, PDFContentScrapingStrategy): + if _needs_pdf_crawler(crawler_config, config_list): if hooks_config: # PDFCrawlerStrategy has no browser page, so hooks can't attach to it raise HTTPException(status_code=400, detail="Hooks are not supported with PDFContentScrapingStrategy") # SSRF protection: vet the PDF download URL and every redirect hop - crawler_config.scraping_strategy.url_validator = validate_url_destination + if isinstance(crawler_config.scraping_strategy, PDFContentScrapingStrategy): + crawler_config.scraping_strategy.url_validator = validate_url_destination # Use PDFCrawlerStrategy when scraping PDFs, as headless Chromium can't render PDFs inline crawler = AsyncWebCrawler(crawler_strategy=PDFCrawlerStrategy()) await crawler.start() @@ -964,7 +1022,7 @@ async def handle_stream_crawl_request( ) results_gen = await crawler.arun_many( urls=urls, - config=crawler_config, + config=config_list or crawler_config, dispatcher=dispatcher ) diff --git a/deploy/docker/server.py b/deploy/docker/server.py index 248155278..a05d8939d 100644 --- a/deploy/docker/server.py +++ b/deploy/docker/server.py @@ -1032,7 +1032,8 @@ async def stream_process(crawl_request: CrawlRequestWithHooks): browser_config=crawl_request.browser_config, crawler_config=crawl_request.crawler_config, config=config, - hooks_config=hooks_config + hooks_config=hooks_config, + crawler_configs=crawl_request.crawler_configs, ) # Add hooks info to response headers if available diff --git a/deploy/docker/tests/test_crawler_configs_routing.py b/deploy/docker/tests/test_crawler_configs_routing.py new file mode 100644 index 000000000..65303b3dc --- /dev/null +++ b/deploy/docker/tests/test_crawler_configs_routing.py @@ -0,0 +1,213 @@ +"""crawler_configs must reach arun_many for one URL and on the streaming path. + +The per-URL config list (#1837) was only applied when a non-streaming request +carried two or more URLs. With one URL the handler called arun(), which takes a +single config, and /crawl/stream never passed the list down, so both requests +ran with crawler_config alone and still answered HTTP 200. + +With the list applied, it also has to decide the crawler: a list of PDF entries +runs on PDFCrawlerStrategy, and one crawler cannot serve a list mixing PDF and +browser entries, so that is refused instead of failing mid-crawl. +""" + +import asyncio + +import pytest +from fastapi import HTTPException + +from crawl4ai import CrawlerRunConfig +from crawl4ai.models import CrawlResult + +PER_URL = [ + { + "type": "CrawlerRunConfig", + "params": {"url_matcher": "*example.com*", "word_count_threshold": 7}, + }, +] + + +class RecordingCrawler: + def __init__(self): + self.calls = [] + + async def arun(self, url, config=None, **kwargs): + self.calls.append(("arun", url, config)) + return CrawlResult(url=url, html="", success=True) + + async def arun_many(self, urls, config=None, **kwargs): + self.calls.append(("arun_many", urls, config)) + return [CrawlResult(url=url, html="", success=True) for url in urls] + + +def _install(monkeypatch, crawler): + import api + import crawler_pool + + async def get_crawler(*args, **kwargs): + return crawler + + async def release_crawler(*args, **kwargs): + return None + + # Offline: the seed check resolves DNS. + monkeypatch.setattr(api, "validate_url_destination", lambda url: None) + monkeypatch.setattr(crawler_pool, "get_crawler", get_crawler) + monkeypatch.setattr(crawler_pool, "release_crawler", release_crawler) + return api + + +def test_single_url_uses_the_config_list(server_module, monkeypatch): + crawler = RecordingCrawler() + api = _install(monkeypatch, crawler) + + asyncio.run( + api.handle_crawl_request( + urls=["https://example.com/"], + browser_config={}, + crawler_config={}, + config=server_module.config, + crawler_configs=PER_URL, + ) + ) + + [(method, urls, config)] = crawler.calls + assert method == "arun_many" + assert urls == ["https://example.com/"] + assert [c.word_count_threshold for c in config] == [7] + + +def test_single_url_without_a_list_still_uses_arun(server_module, monkeypatch): + crawler = RecordingCrawler() + api = _install(monkeypatch, crawler) + + asyncio.run( + api.handle_crawl_request( + urls=["https://example.com/"], + browser_config={}, + crawler_config={}, + config=server_module.config, + ) + ) + + [(method, url, config)] = crawler.calls + assert method == "arun" + assert url == "https://example.com/" + assert isinstance(config, CrawlerRunConfig) + + +def test_stream_passes_the_config_list(server_module, monkeypatch): + crawler = RecordingCrawler() + api = _install(monkeypatch, crawler) + + asyncio.run( + api.handle_stream_crawl_request( + urls=["https://example.com/", "https://example.org/"], + browser_config={}, + crawler_config={}, + config=server_module.config, + crawler_configs=PER_URL, + ) + ) + + [(method, urls, config)] = crawler.calls + assert method == "arun_many" + assert [c.word_count_threshold for c in config] == [7] + # arun_many takes the stream flag from the first config of a list. + assert all(c.stream for c in config) + + +def test_stream_endpoint_forwards_the_list(server_module, monkeypatch): + captured = {} + + async def fake_handle_stream(**kwargs): + captured.update(kwargs) + raise RuntimeError("stop after capture") + + monkeypatch.setattr( + server_module, "handle_stream_crawl_request", fake_handle_stream + ) + request = server_module.CrawlRequestWithHooks( + urls=["https://example.com/"], crawler_configs=PER_URL + ) + + try: + asyncio.run(server_module.stream_process(crawl_request=request)) + except RuntimeError: + pass + + assert captured["crawler_configs"] == PER_URL + + +PDF = {"type": "PDFContentScrapingStrategy", "params": {}} +ALL_PDF = [ + { + "type": "CrawlerRunConfig", + "params": {"url_matcher": "*.pdf", "scraping_strategy": PDF}, + }, +] +MIXED = ALL_PDF + [{"type": "CrawlerRunConfig", "params": {"url_matcher": "*"}}] + + +class RecordingPdfCrawler(RecordingCrawler): + """Stands in for AsyncWebCrawler(crawler_strategy=PDFCrawlerStrategy()).""" + + def __init__(self, crawler_strategy=None, **kwargs): + super().__init__() + self.crawler_strategy = crawler_strategy + + async def start(self): + return self + + async def close(self): + return None + + +def test_a_list_of_pdf_entries_runs_on_the_pdf_crawler(server_module, monkeypatch): + from crawl4ai.processors.pdf import PDFCrawlerStrategy + + api = _install(monkeypatch, RecordingCrawler()) + made = [] + + def pdf_crawler(**kwargs): + made.append(RecordingPdfCrawler(**kwargs)) + return made[-1] + + monkeypatch.setattr(api, "AsyncWebCrawler", pdf_crawler) + + asyncio.run( + api.handle_crawl_request( + urls=["https://example.com/a.pdf"], + browser_config={}, + crawler_config={}, + config=server_module.config, + crawler_configs=ALL_PDF, + ) + ) + + [crawler] = made + assert isinstance(crawler.crawler_strategy, PDFCrawlerStrategy) + [(method, urls, config)] = crawler.calls + assert method == "arun_many" + # SSRF: every per-URL PDF entry still gets the destination check. + assert config[0].scraping_strategy.url_validator is api.validate_url_destination + + +@pytest.mark.parametrize("stream", [False, True]) +def test_a_list_mixing_pdf_and_browser_entries_is_refused( + server_module, monkeypatch, stream +): + api = _install(monkeypatch, RecordingCrawler()) + handler = api.handle_stream_crawl_request if stream else api.handle_crawl_request + + with pytest.raises(HTTPException) as refused: + asyncio.run( + handler( + urls=["https://example.com/a.pdf", "https://example.com/"], + browser_config={}, + crawler_config={}, + config=server_module.config, + crawler_configs=MIXED, + ) + ) + + assert refused.value.status_code == 400 diff --git a/tests/test_issue_1837_config_list.py b/tests/test_issue_1837_config_list.py index 45150e589..0e017be92 100644 --- a/tests/test_issue_1837_config_list.py +++ b/tests/test_issue_1837_config_list.py @@ -88,11 +88,11 @@ class TestConfigListLogic: """Verify the branching logic for single vs list configs.""" def test_api_uses_config_list_when_provided(self): - """When crawler_configs is provided with multiple URLs, it should be used.""" + """When crawler_configs is provided, it should be used, for one URL or many.""" with open("deploy/docker/api.py") as f: source = f.read() # Should check crawler_configs and build a list - assert "if crawler_configs and len(urls) > 1:" in source + assert "_load_crawler_configs(crawler_configs, base_config)" in source assert "config_list" in source def test_api_falls_back_to_single_config(self): @@ -125,12 +125,14 @@ def test_server_passes_crawler_configs(self): class TestBackwardCompatibility: """Ensure existing single-config requests still work.""" - def test_single_url_ignores_crawler_configs(self): - """With a single URL, crawler_configs should be ignored (uses arun, not arun_many).""" + def test_single_url_with_config_list_uses_arun_many(self): + """A config list goes to arun_many even for a single URL (arun takes one config). + + Behavioral coverage: deploy/docker/tests/test_crawler_configs_routing.py. + """ with open("deploy/docker/api.py") as f: source = f.read() - # Single URL uses arun which only takes one config - assert '"arun" if len(urls) == 1 else "arun_many"' in source + assert "use_many = len(urls) > 1 or isinstance(effective_config, list)" in source def test_no_crawler_configs_uses_single(self): """When crawler_configs is None, the original single config path is used."""