From c899d1a1ed0dd45c8dab3fd2345cf1dd7fc73971 Mon Sep 17 00:00:00 2001 From: Talel Boussetta Date: Fri, 25 Sep 2026 10:29:47 +0100 Subject: [PATCH 1/3] fix(docker): apply crawler_configs to single-URL and streaming requests The per-URL config list (#1837) was only honoured when a non-streaming request carried two or more URLs. With one URL the handler called arun(), which takes a single config, and /crawl/stream (and /crawl with stream=true) never passed the list to handle_stream_crawl_request. Both requests ran with crawler_config alone and still answered 200. A request that carries a list now goes to arun_many whatever its URL count, and the streaming handler takes and applies the list. Loading moves into _load_crawler_configs(), shared by both handlers the way _normalize_and_validate_seeds() is, so both apply the same trust boundary and wire the PDF URL validator into every entry. On the streaming path each entry gets stream=True, since arun_many reads the stream flag from the first config of a list. The legacy source-grep test that pinned the single-URL path to arun() is updated to the new routing; behavioural tests are in deploy/docker/tests/test_crawler_configs_routing.py. --- deploy/docker/api.py | 55 +++++--- deploy/docker/server.py | 3 +- .../tests/test_crawler_configs_routing.py | 131 ++++++++++++++++++ tests/test_issue_1837_config_list.py | 14 +- 4 files changed, 180 insertions(+), 23 deletions(-) create mode 100644 deploy/docker/tests/test_crawler_configs_routing.py diff --git a/deploy/docker/api.py b/deploy/docker/api.py index 7c90fb5ba..40743786f 100644 --- a/deploy/docker/api.py +++ b/deploy/docker/api.py @@ -647,6 +647,32 @@ async def stream_results(crawler: AsyncWebCrawler, results_gen: AsyncGenerator) await _dispose_crawler(crawler) +def _load_crawler_configs( + crawler_configs: List[dict], + base_config: Optional[dict] = None, + stream: bool = False, +) -> List[CrawlerRunConfig]: + """Deserialize a per-URL crawler_configs list under the untrusted gate. + Shared by the streaming and non-streaming crawl handlers so both apply the + same boundary and wire the PDF URL validator into every entry.""" + from crawl4ai.processors.pdf import PDFContentScrapingStrategy + + config_list = [CrawlerRunConfig.load(cc, provenance=Provenance.UNTRUSTED) for cc in crawler_configs] + for cfg in config_list: + for key, value in (base_config or {}).items(): + if hasattr(cfg, key): + current_value = getattr(cfg, key) + if current_value is None or current_value == "": + setattr(cfg, key, value) + # SSRF: per-URL PDF strategies need the validator wired too + if isinstance(cfg.scraping_strategy, PDFContentScrapingStrategy): + cfg.scraping_strategy.url_validator = validate_url_destination + if stream: + # arun_many takes the stream flag from the first config of a list. + cfg.stream = True + return config_list + + def _normalize_and_validate_seeds(urls: List[str]) -> List[str]: """Prefix bare hosts with https:// and SSRF-validate every seed URL's destination. Shared by the streaming and non-streaming crawl handlers so a @@ -723,19 +749,9 @@ async def handle_crawl_request( base_config = config["crawler"]["base_config"] # Build the config(s) to pass to arun/arun_many - if crawler_configs and len(urls) > 1: + if crawler_configs: # Per-URL config list: deserialize each and apply base_config - config_list = [CrawlerRunConfig.load(cc, provenance=Provenance.UNTRUSTED) for cc in crawler_configs] - for cfg in config_list: - for key, value in base_config.items(): - if hasattr(cfg, key): - current_value = getattr(cfg, key) - if current_value is None or current_value == "": - setattr(cfg, key, value) - # SSRF: per-URL PDF strategies need the validator wired too - if isinstance(cfg.scraping_strategy, PDFContentScrapingStrategy): - cfg.scraping_strategy.url_validator = validate_url_destination - effective_config = config_list + effective_config = _load_crawler_configs(crawler_configs, base_config) else: # Single config (original behavior) for key, value in base_config.items(): @@ -746,9 +762,12 @@ async def handle_crawl_request( effective_config = crawler_config results = [] - func = getattr(crawler, "arun" if len(urls) == 1 else "arun_many") + # arun takes one config, so a per-URL list goes to arun_many even for + # a single URL; arun_many matches each URL against the list. + use_many = len(urls) > 1 or isinstance(effective_config, list) + func = getattr(crawler, "arun_many" if use_many else "arun") partial_func = partial(func, - urls[0] if len(urls) == 1 else urls, + urls if use_many else urls[0], config=effective_config, dispatcher=dispatcher) # Optional per-crawl wall-clock deadline (config limits.wall_clock_s; 0 = none). @@ -891,7 +910,8 @@ async def handle_stream_crawl_request( browser_config: dict, crawler_config: dict, config: dict, - hooks_config: Optional[dict] = None + hooks_config: Optional[dict] = None, + crawler_configs: Optional[List[dict]] = None, ) -> Tuple[AsyncWebCrawler, AsyncGenerator, Optional[Dict]]: """Handle streaming crawl requests with optional hooks.""" hooks_info = None @@ -916,6 +936,9 @@ async def handle_stream_crawl_request( clamp_deep_crawl(crawler_config) crawler_config.stream = True + config_list = ( + _load_crawler_configs(crawler_configs, stream=True) if crawler_configs else None + ) # Deep crawl streaming supports exactly one start URL if crawler_config.deep_crawl_strategy is not None and len(urls) != 1: @@ -964,7 +987,7 @@ async def handle_stream_crawl_request( ) results_gen = await crawler.arun_many( urls=urls, - config=crawler_config, + config=config_list or crawler_config, dispatcher=dispatcher ) diff --git a/deploy/docker/server.py b/deploy/docker/server.py index 248155278..a05d8939d 100644 --- a/deploy/docker/server.py +++ b/deploy/docker/server.py @@ -1032,7 +1032,8 @@ async def stream_process(crawl_request: CrawlRequestWithHooks): browser_config=crawl_request.browser_config, crawler_config=crawl_request.crawler_config, config=config, - hooks_config=hooks_config + hooks_config=hooks_config, + crawler_configs=crawl_request.crawler_configs, ) # Add hooks info to response headers if available diff --git a/deploy/docker/tests/test_crawler_configs_routing.py b/deploy/docker/tests/test_crawler_configs_routing.py new file mode 100644 index 000000000..b97093dfc --- /dev/null +++ b/deploy/docker/tests/test_crawler_configs_routing.py @@ -0,0 +1,131 @@ +"""crawler_configs must reach arun_many for one URL and on the streaming path. + +The per-URL config list (#1837) was only applied when a non-streaming request +carried two or more URLs. With one URL the handler called arun(), which takes a +single config, and /crawl/stream never passed the list down, so both requests +ran with crawler_config alone and still answered HTTP 200. +""" + +import asyncio + +from crawl4ai import CrawlerRunConfig +from crawl4ai.models import CrawlResult + +PER_URL = [ + { + "type": "CrawlerRunConfig", + "params": {"url_matcher": "*example.com*", "word_count_threshold": 7}, + }, +] + + +class RecordingCrawler: + def __init__(self): + self.calls = [] + + async def arun(self, url, config=None, **kwargs): + self.calls.append(("arun", url, config)) + return CrawlResult(url=url, html="", success=True) + + async def arun_many(self, urls, config=None, **kwargs): + self.calls.append(("arun_many", urls, config)) + return [CrawlResult(url=url, html="", success=True) for url in urls] + + +def _install(monkeypatch, crawler): + import api + import crawler_pool + + async def get_crawler(*args, **kwargs): + return crawler + + async def release_crawler(*args, **kwargs): + return None + + # Offline: the seed check resolves DNS. + monkeypatch.setattr(api, "validate_url_destination", lambda url: None) + monkeypatch.setattr(crawler_pool, "get_crawler", get_crawler) + monkeypatch.setattr(crawler_pool, "release_crawler", release_crawler) + return api + + +def test_single_url_uses_the_config_list(server_module, monkeypatch): + crawler = RecordingCrawler() + api = _install(monkeypatch, crawler) + + asyncio.run( + api.handle_crawl_request( + urls=["https://example.com/"], + browser_config={}, + crawler_config={}, + config=server_module.config, + crawler_configs=PER_URL, + ) + ) + + [(method, urls, config)] = crawler.calls + assert method == "arun_many" + assert urls == ["https://example.com/"] + assert [c.word_count_threshold for c in config] == [7] + + +def test_single_url_without_a_list_still_uses_arun(server_module, monkeypatch): + crawler = RecordingCrawler() + api = _install(monkeypatch, crawler) + + asyncio.run( + api.handle_crawl_request( + urls=["https://example.com/"], + browser_config={}, + crawler_config={}, + config=server_module.config, + ) + ) + + [(method, url, config)] = crawler.calls + assert method == "arun" + assert url == "https://example.com/" + assert isinstance(config, CrawlerRunConfig) + + +def test_stream_passes_the_config_list(server_module, monkeypatch): + crawler = RecordingCrawler() + api = _install(monkeypatch, crawler) + + asyncio.run( + api.handle_stream_crawl_request( + urls=["https://example.com/", "https://example.org/"], + browser_config={}, + crawler_config={}, + config=server_module.config, + crawler_configs=PER_URL, + ) + ) + + [(method, urls, config)] = crawler.calls + assert method == "arun_many" + assert [c.word_count_threshold for c in config] == [7] + # arun_many takes the stream flag from the first config of a list. + assert all(c.stream for c in config) + + +def test_stream_endpoint_forwards_the_list(server_module, monkeypatch): + captured = {} + + async def fake_handle_stream(**kwargs): + captured.update(kwargs) + raise RuntimeError("stop after capture") + + monkeypatch.setattr( + server_module, "handle_stream_crawl_request", fake_handle_stream + ) + request = server_module.CrawlRequestWithHooks( + urls=["https://example.com/"], crawler_configs=PER_URL + ) + + try: + asyncio.run(server_module.stream_process(crawl_request=request)) + except RuntimeError: + pass + + assert captured["crawler_configs"] == PER_URL diff --git a/tests/test_issue_1837_config_list.py b/tests/test_issue_1837_config_list.py index 45150e589..4a888e070 100644 --- a/tests/test_issue_1837_config_list.py +++ b/tests/test_issue_1837_config_list.py @@ -88,11 +88,11 @@ class TestConfigListLogic: """Verify the branching logic for single vs list configs.""" def test_api_uses_config_list_when_provided(self): - """When crawler_configs is provided with multiple URLs, it should be used.""" + """When crawler_configs is provided, it should be used, for one URL or many.""" with open("deploy/docker/api.py") as f: source = f.read() # Should check crawler_configs and build a list - assert "if crawler_configs and len(urls) > 1:" in source + assert "if crawler_configs:" in source assert "config_list" in source def test_api_falls_back_to_single_config(self): @@ -125,12 +125,14 @@ def test_server_passes_crawler_configs(self): class TestBackwardCompatibility: """Ensure existing single-config requests still work.""" - def test_single_url_ignores_crawler_configs(self): - """With a single URL, crawler_configs should be ignored (uses arun, not arun_many).""" + def test_single_url_with_config_list_uses_arun_many(self): + """A config list goes to arun_many even for a single URL (arun takes one config). + + Behavioral coverage: deploy/docker/tests/test_crawler_configs_routing.py. + """ with open("deploy/docker/api.py") as f: source = f.read() - # Single URL uses arun which only takes one config - assert '"arun" if len(urls) == 1 else "arun_many"' in source + assert "use_many = len(urls) > 1 or isinstance(effective_config, list)" in source def test_no_crawler_configs_uses_single(self): """When crawler_configs is None, the original single config path is used.""" From 5ae28573d3087855d6abd633be58ba83fdae0ff7 Mon Sep 17 00:00:00 2001 From: Talel Boussetta Date: Fri, 25 Sep 2026 11:28:36 +0100 Subject: [PATCH 2/3] fix(docker): let a per-URL config list decide the PDF crawler With crawler_configs applied, the crawler was still chosen from the top-level crawler_config alone, so a list entry using PDFContentScrapingStrategy ran on the pooled browser crawler and failed before the PDF scraper could run. _needs_pdf_crawler() decides from the list when there is one: a list of PDF entries runs on PDFCrawlerStrategy, and since one crawler serves every URL of a request, a list mixing PDF and browser entries is refused with 400 instead of failing mid-crawl. The non-streaming handler now loads the list before it picks a crawler, as the streaming one already did. --- deploy/docker/api.py | 44 +++++++--- .../tests/test_crawler_configs_routing.py | 82 +++++++++++++++++++ tests/test_issue_1837_config_list.py | 2 +- 3 files changed, 117 insertions(+), 11 deletions(-) diff --git a/deploy/docker/api.py b/deploy/docker/api.py index 40743786f..d9421d748 100644 --- a/deploy/docker/api.py +++ b/deploy/docker/api.py @@ -673,6 +673,26 @@ def _load_crawler_configs( return config_list +def _needs_pdf_crawler( + crawler_config: CrawlerRunConfig, + config_list: Optional[List[CrawlerRunConfig]] = None, +) -> bool: + """Whether the crawl runs on PDFCrawlerStrategy. A per-URL list decides when + present, since it is what arun_many runs. One crawler serves every URL of a + request, so a list mixing PDF and browser entries is refused rather than run + on the wrong one.""" + from crawl4ai.processors.pdf import PDFContentScrapingStrategy + + configs = config_list or [crawler_config] + pdf = [isinstance(cfg.scraping_strategy, PDFContentScrapingStrategy) for cfg in configs] + if any(pdf) and not all(pdf): + raise HTTPException( + status_code=400, + detail="crawler_configs cannot mix PDFContentScrapingStrategy with other scraping strategies in one request", + ) + return all(pdf) + + def _normalize_and_validate_seeds(urls: List[str]) -> List[str]: """Prefix bare hosts with https:// and SSRF-validate every seed URL's destination. Shared by the streaming and non-streaming crawl handlers so a @@ -725,33 +745,36 @@ async def handle_crawl_request( ) if config["crawler"]["rate_limiter"]["enabled"] else None ) + base_config = config["crawler"]["base_config"] + # Per-URL config list: deserialize each and apply base_config. Loaded + # before a crawler is chosen, because the list decides which one runs. + config_list = _load_crawler_configs(crawler_configs, base_config) if crawler_configs else None + from crawler_pool import get_crawler, release_crawler from crawl4ai.processors.pdf import PDFContentScrapingStrategy, PDFCrawlerStrategy - is_pdf_crawl = isinstance(crawler_config.scraping_strategy, PDFContentScrapingStrategy) + is_pdf_crawl = _needs_pdf_crawler(crawler_config, config_list) if is_pdf_crawl: if hooks_config: # PDFCrawlerStrategy has no browser page, so hooks can't attach to it raise HTTPException(status_code=400, detail="Hooks are not supported with PDFContentScrapingStrategy") # SSRF protection: vet the PDF download URL and every redirect hop - crawler_config.scraping_strategy.url_validator = validate_url_destination + if isinstance(crawler_config.scraping_strategy, PDFContentScrapingStrategy): + crawler_config.scraping_strategy.url_validator = validate_url_destination # Use PDFCrawlerStrategy when scraping PDFs, as headless Chromium can't render PDFs inline crawler = AsyncWebCrawler(crawler_strategy=PDFCrawlerStrategy()) await crawler.start() else: crawler = await get_crawler(browser_config) - + # Attach declarative hooks if provided hooks_status = {} if hooks_config: hooks_status = _attach_declarative_hooks(crawler, hooks_config) logger.info(f"Hooks attachment status: {hooks_status['status']}") - - base_config = config["crawler"]["base_config"] # Build the config(s) to pass to arun/arun_many - if crawler_configs: - # Per-URL config list: deserialize each and apply base_config - effective_config = _load_crawler_configs(crawler_configs, base_config) + if config_list: + effective_config = config_list else: # Single config (original behavior) for key, value in base_config.items(): @@ -952,12 +975,13 @@ async def handle_stream_crawl_request( from crawler_pool import get_crawler from crawl4ai.processors.pdf import PDFContentScrapingStrategy, PDFCrawlerStrategy - if isinstance(crawler_config.scraping_strategy, PDFContentScrapingStrategy): + if _needs_pdf_crawler(crawler_config, config_list): if hooks_config: # PDFCrawlerStrategy has no browser page, so hooks can't attach to it raise HTTPException(status_code=400, detail="Hooks are not supported with PDFContentScrapingStrategy") # SSRF protection: vet the PDF download URL and every redirect hop - crawler_config.scraping_strategy.url_validator = validate_url_destination + if isinstance(crawler_config.scraping_strategy, PDFContentScrapingStrategy): + crawler_config.scraping_strategy.url_validator = validate_url_destination # Use PDFCrawlerStrategy when scraping PDFs, as headless Chromium can't render PDFs inline crawler = AsyncWebCrawler(crawler_strategy=PDFCrawlerStrategy()) await crawler.start() diff --git a/deploy/docker/tests/test_crawler_configs_routing.py b/deploy/docker/tests/test_crawler_configs_routing.py index b97093dfc..65303b3dc 100644 --- a/deploy/docker/tests/test_crawler_configs_routing.py +++ b/deploy/docker/tests/test_crawler_configs_routing.py @@ -4,10 +4,17 @@ carried two or more URLs. With one URL the handler called arun(), which takes a single config, and /crawl/stream never passed the list down, so both requests ran with crawler_config alone and still answered HTTP 200. + +With the list applied, it also has to decide the crawler: a list of PDF entries +runs on PDFCrawlerStrategy, and one crawler cannot serve a list mixing PDF and +browser entries, so that is refused instead of failing mid-crawl. """ import asyncio +import pytest +from fastapi import HTTPException + from crawl4ai import CrawlerRunConfig from crawl4ai.models import CrawlResult @@ -129,3 +136,78 @@ async def fake_handle_stream(**kwargs): pass assert captured["crawler_configs"] == PER_URL + + +PDF = {"type": "PDFContentScrapingStrategy", "params": {}} +ALL_PDF = [ + { + "type": "CrawlerRunConfig", + "params": {"url_matcher": "*.pdf", "scraping_strategy": PDF}, + }, +] +MIXED = ALL_PDF + [{"type": "CrawlerRunConfig", "params": {"url_matcher": "*"}}] + + +class RecordingPdfCrawler(RecordingCrawler): + """Stands in for AsyncWebCrawler(crawler_strategy=PDFCrawlerStrategy()).""" + + def __init__(self, crawler_strategy=None, **kwargs): + super().__init__() + self.crawler_strategy = crawler_strategy + + async def start(self): + return self + + async def close(self): + return None + + +def test_a_list_of_pdf_entries_runs_on_the_pdf_crawler(server_module, monkeypatch): + from crawl4ai.processors.pdf import PDFCrawlerStrategy + + api = _install(monkeypatch, RecordingCrawler()) + made = [] + + def pdf_crawler(**kwargs): + made.append(RecordingPdfCrawler(**kwargs)) + return made[-1] + + monkeypatch.setattr(api, "AsyncWebCrawler", pdf_crawler) + + asyncio.run( + api.handle_crawl_request( + urls=["https://example.com/a.pdf"], + browser_config={}, + crawler_config={}, + config=server_module.config, + crawler_configs=ALL_PDF, + ) + ) + + [crawler] = made + assert isinstance(crawler.crawler_strategy, PDFCrawlerStrategy) + [(method, urls, config)] = crawler.calls + assert method == "arun_many" + # SSRF: every per-URL PDF entry still gets the destination check. + assert config[0].scraping_strategy.url_validator is api.validate_url_destination + + +@pytest.mark.parametrize("stream", [False, True]) +def test_a_list_mixing_pdf_and_browser_entries_is_refused( + server_module, monkeypatch, stream +): + api = _install(monkeypatch, RecordingCrawler()) + handler = api.handle_stream_crawl_request if stream else api.handle_crawl_request + + with pytest.raises(HTTPException) as refused: + asyncio.run( + handler( + urls=["https://example.com/a.pdf", "https://example.com/"], + browser_config={}, + crawler_config={}, + config=server_module.config, + crawler_configs=MIXED, + ) + ) + + assert refused.value.status_code == 400 diff --git a/tests/test_issue_1837_config_list.py b/tests/test_issue_1837_config_list.py index 4a888e070..0e5463bba 100644 --- a/tests/test_issue_1837_config_list.py +++ b/tests/test_issue_1837_config_list.py @@ -92,7 +92,7 @@ def test_api_uses_config_list_when_provided(self): with open("deploy/docker/api.py") as f: source = f.read() # Should check crawler_configs and build a list - assert "if crawler_configs:" in source + assert "_load_crawler_configs(crawler_configs, base_config) if crawler_configs else None" in source assert "config_list" in source def test_api_falls_back_to_single_config(self): From 632f0bc3d9ae1f207003cbc752e58fa8e18e9855 Mon Sep 17 00:00:00 2001 From: Talel Boussetta Date: Sat, 26 Sep 2026 16:59:11 +0100 Subject: [PATCH 3/3] style(docker): black-format the new crawler_configs code Wraps the new lines over black's limit. The source assertion in test_issue_1837_config_list.py now matches the call rather than the whole one-line conditional, so it no longer depends on how the line is wrapped. No behaviour change. --- deploy/docker/api.py | 19 +++++++++++++++---- tests/test_issue_1837_config_list.py | 2 +- 2 files changed, 16 insertions(+), 5 deletions(-) diff --git a/deploy/docker/api.py b/deploy/docker/api.py index d9421d748..20ab6d508 100644 --- a/deploy/docker/api.py +++ b/deploy/docker/api.py @@ -657,7 +657,10 @@ def _load_crawler_configs( same boundary and wire the PDF URL validator into every entry.""" from crawl4ai.processors.pdf import PDFContentScrapingStrategy - config_list = [CrawlerRunConfig.load(cc, provenance=Provenance.UNTRUSTED) for cc in crawler_configs] + config_list = [ + CrawlerRunConfig.load(cc, provenance=Provenance.UNTRUSTED) + for cc in crawler_configs + ] for cfg in config_list: for key, value in (base_config or {}).items(): if hasattr(cfg, key): @@ -684,7 +687,9 @@ def _needs_pdf_crawler( from crawl4ai.processors.pdf import PDFContentScrapingStrategy configs = config_list or [crawler_config] - pdf = [isinstance(cfg.scraping_strategy, PDFContentScrapingStrategy) for cfg in configs] + pdf = [ + isinstance(cfg.scraping_strategy, PDFContentScrapingStrategy) for cfg in configs + ] if any(pdf) and not all(pdf): raise HTTPException( status_code=400, @@ -748,7 +753,11 @@ async def handle_crawl_request( base_config = config["crawler"]["base_config"] # Per-URL config list: deserialize each and apply base_config. Loaded # before a crawler is chosen, because the list decides which one runs. - config_list = _load_crawler_configs(crawler_configs, base_config) if crawler_configs else None + config_list = ( + _load_crawler_configs(crawler_configs, base_config) + if crawler_configs + else None + ) from crawler_pool import get_crawler, release_crawler from crawl4ai.processors.pdf import PDFContentScrapingStrategy, PDFCrawlerStrategy @@ -960,7 +969,9 @@ async def handle_stream_crawl_request( clamp_deep_crawl(crawler_config) crawler_config.stream = True config_list = ( - _load_crawler_configs(crawler_configs, stream=True) if crawler_configs else None + _load_crawler_configs(crawler_configs, stream=True) + if crawler_configs + else None ) # Deep crawl streaming supports exactly one start URL diff --git a/tests/test_issue_1837_config_list.py b/tests/test_issue_1837_config_list.py index 0e5463bba..0e017be92 100644 --- a/tests/test_issue_1837_config_list.py +++ b/tests/test_issue_1837_config_list.py @@ -92,7 +92,7 @@ def test_api_uses_config_list_when_provided(self): with open("deploy/docker/api.py") as f: source = f.read() # Should check crawler_configs and build a list - assert "_load_crawler_configs(crawler_configs, base_config) if crawler_configs else None" in source + assert "_load_crawler_configs(crawler_configs, base_config)" in source assert "config_list" in source def test_api_falls_back_to_single_config(self):