From 77d5aa6165ff5d08903421afe1d5245d913a82d2 Mon Sep 17 00:00:00 2001
From: ktz03 <2484593937@qq.com>
Date: Fri, 9 Oct 2026 14:51:41 +0800
Subject: [PATCH] fix(markdown): resolve relative HTML base URLs
---
crawl4ai/html2text/__init__.py | 7 +---
tests/unit/test_markdown_base_url.py | 57 ++++++++++++++++++++++++++++
2 files changed, 59 insertions(+), 5 deletions(-)
create mode 100644 tests/unit/test_markdown_base_url.py
diff --git a/crawl4ai/html2text/__init__.py b/crawl4ai/html2text/__init__.py
index ff39fb211..b3079db17 100644
--- a/crawl4ai/html2text/__init__.py
+++ b/crawl4ai/html2text/__init__.py
@@ -320,7 +320,7 @@ def handle_tag(
if tag == "base" and start:
href = attrs.get("href")
if href:
- self.baseurl = href
+ self.baseurl = urlparse.urljoin(self.baseurl, href)
# first thing inside the anchor tag is another tag
# that produces some output
@@ -1107,10 +1107,7 @@ def handle_tag(self, tag, attrs, start):
# Handle tag to update base URL for relative links
# Must be handled before preserved tags since is in
if tag == "base" and start:
- href = attrs.get("href") if attrs else None
- if href:
- self.baseurl = href
- # Also let parent class handle it
+ # Let the parent resolve relative hrefs against the document URL once.
return super().handle_tag(tag, attrs, start)
# Handle preserved tags
diff --git a/tests/unit/test_markdown_base_url.py b/tests/unit/test_markdown_base_url.py
new file mode 100644
index 000000000..b0a6159eb
--- /dev/null
+++ b/tests/unit/test_markdown_base_url.py
@@ -0,0 +1,57 @@
+"""Relative base elements resolve against the original document URL."""
+
+import pytest
+
+from crawl4ai import DefaultMarkdownGenerator
+from crawl4ai.html2text import CustomHTML2Text, HTML2Text
+
+BASE_CASES = [
+ ("../assets/", "https://example.test/assets/"),
+ ("assets/", "https://example.test/pages/assets/"),
+ ("/docs/", "https://example.test/docs/"),
+ ("//cdn.example.test/v1/", "https://cdn.example.test/v1/"),
+ ("https://cdn.example.test/v1/", "https://cdn.example.test/v1/"),
+ (None, "https://example.test/pages/"),
+ ("", "https://example.test/pages/"),
+]
+PAGE_URL = "https://example.test/pages/index.html"
+
+
+def _page_html(base_href):
+ base_tag = f'' if base_href is not None else ""
+ return (
+ f"{base_tag}"
+ 'Guide
'
+ "
"
+ )
+
+
+@pytest.mark.parametrize("converter_class", [HTML2Text, CustomHTML2Text])
+@pytest.mark.parametrize(("base_href", "expected_prefix"), BASE_CASES)
+def test_relative_base_element_resolves_links_and_images(
+ converter_class, base_href, expected_prefix
+):
+ converter = converter_class(baseurl=PAGE_URL, bodywidth=0)
+
+ markdown = converter.handle(_page_html(base_href))
+
+ assert f"[Guide]({expected_prefix}guide.html)" in markdown
+ assert f"" in markdown
+
+
+@pytest.mark.parametrize("citations", [False, True])
+@pytest.mark.parametrize(("base_href", "expected_prefix"), BASE_CASES)
+def test_markdown_generator_resolves_relative_base_element(
+ citations, base_href, expected_prefix
+):
+ result = DefaultMarkdownGenerator().generate_markdown(
+ _page_html(base_href), base_url=PAGE_URL, citations=citations
+ )
+
+ assert f"[Guide]({expected_prefix}guide.html)" in result.raw_markdown
+ assert f"" in result.raw_markdown
+ if citations:
+ assert f"{expected_prefix}guide.html" in result.references_markdown
+ assert f"{expected_prefix}logo.png" in result.references_markdown
+ else:
+ assert result.references_markdown == ""