From 77d5aa6165ff5d08903421afe1d5245d913a82d2 Mon Sep 17 00:00:00 2001 From: ktz03 <2484593937@qq.com> Date: Fri, 9 Oct 2026 14:51:41 +0800 Subject: [PATCH] fix(markdown): resolve relative HTML base URLs --- crawl4ai/html2text/__init__.py | 7 +--- tests/unit/test_markdown_base_url.py | 57 ++++++++++++++++++++++++++++ 2 files changed, 59 insertions(+), 5 deletions(-) create mode 100644 tests/unit/test_markdown_base_url.py diff --git a/crawl4ai/html2text/__init__.py b/crawl4ai/html2text/__init__.py index ff39fb211..b3079db17 100644 --- a/crawl4ai/html2text/__init__.py +++ b/crawl4ai/html2text/__init__.py @@ -320,7 +320,7 @@ def handle_tag( if tag == "base" and start: href = attrs.get("href") if href: - self.baseurl = href + self.baseurl = urlparse.urljoin(self.baseurl, href) # first thing inside the anchor tag is another tag # that produces some output @@ -1107,10 +1107,7 @@ def handle_tag(self, tag, attrs, start): # Handle tag to update base URL for relative links # Must be handled before preserved tags since is in if tag == "base" and start: - href = attrs.get("href") if attrs else None - if href: - self.baseurl = href - # Also let parent class handle it + # Let the parent resolve relative hrefs against the document URL once. return super().handle_tag(tag, attrs, start) # Handle preserved tags diff --git a/tests/unit/test_markdown_base_url.py b/tests/unit/test_markdown_base_url.py new file mode 100644 index 000000000..b0a6159eb --- /dev/null +++ b/tests/unit/test_markdown_base_url.py @@ -0,0 +1,57 @@ +"""Relative base elements resolve against the original document URL.""" + +import pytest + +from crawl4ai import DefaultMarkdownGenerator +from crawl4ai.html2text import CustomHTML2Text, HTML2Text + +BASE_CASES = [ + ("../assets/", "https://example.test/assets/"), + ("assets/", "https://example.test/pages/assets/"), + ("/docs/", "https://example.test/docs/"), + ("//cdn.example.test/v1/", "https://cdn.example.test/v1/"), + ("https://cdn.example.test/v1/", "https://cdn.example.test/v1/"), + (None, "https://example.test/pages/"), + ("", "https://example.test/pages/"), +] +PAGE_URL = "https://example.test/pages/index.html" + + +def _page_html(base_href): + base_tag = f'' if base_href is not None else "" + return ( + f"{base_tag}

" + 'GuideLogo' + "

" + ) + + +@pytest.mark.parametrize("converter_class", [HTML2Text, CustomHTML2Text]) +@pytest.mark.parametrize(("base_href", "expected_prefix"), BASE_CASES) +def test_relative_base_element_resolves_links_and_images( + converter_class, base_href, expected_prefix +): + converter = converter_class(baseurl=PAGE_URL, bodywidth=0) + + markdown = converter.handle(_page_html(base_href)) + + assert f"[Guide]({expected_prefix}guide.html)" in markdown + assert f"![Logo]({expected_prefix}logo.png)" in markdown + + +@pytest.mark.parametrize("citations", [False, True]) +@pytest.mark.parametrize(("base_href", "expected_prefix"), BASE_CASES) +def test_markdown_generator_resolves_relative_base_element( + citations, base_href, expected_prefix +): + result = DefaultMarkdownGenerator().generate_markdown( + _page_html(base_href), base_url=PAGE_URL, citations=citations + ) + + assert f"[Guide]({expected_prefix}guide.html)" in result.raw_markdown + assert f"![Logo]({expected_prefix}logo.png)" in result.raw_markdown + if citations: + assert f"{expected_prefix}guide.html" in result.references_markdown + assert f"{expected_prefix}logo.png" in result.references_markdown + else: + assert result.references_markdown == ""