diff --git a/common/data_source/html_utils.py b/common/data_source/html_utils.py
index b39569e664..9fd33a7b0e 100644
--- a/common/data_source/html_utils.py
+++ b/common/data_source/html_utils.py
@@ -76,12 +76,24 @@ def format_document_soup(document: bs4.BeautifulSoup, table_cell_separator: str
text = ""
list_element_start = False
verbatim_output = 0
- in_table = False
last_added_newline = False
- link_href: str | None = None
+
+ # ``descendants`` yields opening tags only, so a flag set on
/ would
+ # never clear. Precompute scope by registering each / descendant;
+ # this avoids O(depth) ancestor walks while keeping lookups O(1).
+ table_scope = {id(d) for table in document.find_all("table") for d in table.descendants}
+ href_scope = {}
+ for anchor in document.find_all("a"): # document order, so a nested wins over its parent
+ href_value = anchor.get("href", None)
+ # mostly for typing, having multiple hrefs is not valid HTML
+ link_href = href_value[0] if isinstance(href_value, list) else href_value
+ href_scope.update((id(d), link_href) for d in anchor.descendants)
for e in document.descendants:
verbatim_output -= 1
+ in_table = id(e) in table_scope
+ link_href = href_scope.get(id(e))
+
if isinstance(e, bs4.element.NavigableString):
if isinstance(e, (bs4.element.Comment, bs4.element.Doctype)):
continue
@@ -109,26 +121,15 @@ def format_document_soup(document: bs4.BeautifulSoup, table_cell_separator: str
list_element_start = False
elif isinstance(e, bs4.element.Tag):
- # table is standard HTML element
- if e.name == "table":
- in_table = True
# TR is for rows
- elif e.name == "tr" and in_table:
+ if e.name == "tr" and in_table:
text += "\n"
# td for data cell, th for header
elif e.name in ["td", "th"] and in_table:
text += table_cell_separator
- elif e.name == "/table":
- in_table = False
elif in_table:
# don't handle other cases while in table
pass
- elif e.name == "a":
- href_value = e.get("href", None)
- # mostly for typing, having multiple hrefs is not valid HTML
- link_href = href_value[0] if isinstance(href_value, list) else href_value
- elif e.name == "/a":
- link_href = None
elif e.name in ["p", "div"]:
if not list_element_start:
text += "\n"
diff --git a/test/unit_test/data_source/test_html_utils.py b/test/unit_test/data_source/test_html_utils.py
new file mode 100644
index 0000000000..242ae90d3b
--- /dev/null
+++ b/test/unit_test/data_source/test_html_utils.py
@@ -0,0 +1,91 @@
+#
+# Copyright 2025 The InfiniFlow Authors. All Rights Reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+#
+import bs4
+import pytest
+
+from common.data_source import html_utils
+from common.data_source.config import HtmlBasedConnectorTransformLinksStrategy
+from common.data_source.html_utils import format_document_soup
+
+
+def _fmt(html: str) -> str:
+ return format_document_soup(bs4.BeautifulSoup(html, "html.parser"))
+
+
+@pytest.fixture
+def markdown_links(monkeypatch):
+ """``format_element_text`` only renders links under the markdown strategy."""
+ monkeypatch.setattr(
+ html_utils,
+ "HTML_BASED_CONNECTOR_TRANSFORM_LINKS_STRATEGY",
+ HtmlBasedConnectorTransformLinksStrategy.MARKDOWN,
+ )
+
+
+TABLE = ""
+
+
+def test_paragraph_after_table_keeps_its_newline():
+ assert "\nAfter" in _fmt(f"Before
{TABLE}After
")
+
+
+def test_list_after_table_keeps_hyphen_markers():
+ assert "\n- item1" in _fmt(f"Before
{TABLE}")
+
+
+def test_block_after_table_starts_on_a_new_line():
+ # The div must not be folded onto the last table row.
+ assert "\nTrailing" in _fmt(f"{TABLE}Trailing
")
+
+
+def test_table_still_separates_rows_and_cells():
+ # Control: the table itself must keep working — rows on newlines, cells tab-separated.
+ assert _fmt("") == "A\tB\n\tC\tD"
+
+
+def test_content_after_table_matches_the_same_content_without_a_table(markdown_links):
+ tail = "After
"
+ with_table = _fmt(f"{TABLE}{tail}")
+ without_table = _fmt(tail)
+ assert with_table.endswith(without_table.lstrip("\n"))
+
+
+def test_text_after_link_in_same_paragraph_is_not_linkified(markdown_links):
+ assert _fmt('see link after
') == "see [link](http://x.com) after"
+
+
+def test_paragraph_after_link_is_not_linkified(markdown_links):
+ out = _fmt('link
next paragraph
')
+ assert "[next paragraph]" not in out
+
+
+def test_link_inside_anchor_is_still_linkified(markdown_links):
+ # The anchor text itself must keep its markdown link.
+ assert _fmt('link
') == "[link](http://x.com)"
+
+
+def test_table_cells_do_not_inherit_a_preceding_links_href(markdown_links):
+ out = _fmt(f'pre
{TABLE}')
+ assert "[A](http://x.com)" not in out
+
+
+def test_link_inside_a_table_cell_is_linkified(markdown_links):
+ assert _fmt('') == "[cell](http://x.com)"
+
+
+def test_link_is_stripped_under_the_default_strategy():
+ # Default strategy is STRIP: no markdown link syntax at all.
+ assert _fmt('see link after
') == "see link after"