From d43ba8820b7687dbe99d3887f52a24227356ac8d Mon Sep 17 00:00:00 2001 From: myhloli Date: Fri, 28 Aug 2026 10:58:22 +0800 Subject: [PATCH] fix: isolate HTML styles and pruning metrics --- mineru/model/flash/html/document.py | 9 ++++ mineru/model/flash/html/selector.py | 67 ++++++++++++++++++++++++----- tests/unittest/test_flash_html.py | 32 ++++++++++++++ 3 files changed, 97 insertions(+), 11 deletions(-) diff --git a/mineru/model/flash/html/document.py b/mineru/model/flash/html/document.py index 117395e4..3bb94468 100644 --- a/mineru/model/flash/html/document.py +++ b/mineru/model/flash/html/document.py @@ -105,8 +105,12 @@ def parse_html_document(file_bytes: bytes, source_context: HtmlSourceContext | N continue name = local_name(element) if name == "style": + if _has_discarded_active_ancestor(element): + continue stylesheets.append(HtmlStylesheetSource("inline", "".join(element.itertext()))) elif name == "link" and "stylesheet" in (element.get("rel") or "").casefold().split(): + if _has_discarded_active_ancestor(element): + continue if href := (element.get("href") or "").strip(): stylesheets.append(HtmlStylesheetSource("link", href)) base_href = next( @@ -237,6 +241,11 @@ def _collapsed_text(element: etree._Element) -> str: return re.sub(r"\s+", " ", "".join(element.itertext())).strip() +def _has_discarded_active_ancestor(element: etree._Element) -> bool: + """判断元素是否位于稍后会整棵删除的活动内容祖先中。""" + return any(isinstance(ancestor.tag, str) and local_name(ancestor) in _ACTIVE_TAGS for ancestor in element.iterancestors()) + + def _normalize_formula_sources(root: etree._Element) -> None: """按共享优先级把成功来源收敛为携带裸 LaTeX 的静态 math 元素。""" _preserve_asciimath_text(root) diff --git a/mineru/model/flash/html/selector.py b/mineru/model/flash/html/selector.py index bc72b7cd..9a9c2d71 100644 --- a/mineru/model/flash/html/selector.py +++ b/mineru/model/flash/html/selector.py @@ -86,6 +86,17 @@ class _ScoredCandidate: explicit: bool +@dataclass(frozen=True, slots=True) +class _SoftPruneMetrics: + """保存 soft prune 所需的规范文本段、链接文本和语义对象统计。""" + + text_chars: int = 0 + text_segments: int = 0 + link_chars: int = 0 + link_count: int = 0 + semantic_count: int = 0 + + def select_auto_content(body: etree._Element, stylesheet: MarkupStylesheet) -> ContentSelection: """高置信选择正文候选,任一保守门槛失败时回退完整 body。""" metrics_by_element: dict[etree._Element, CandidateMetrics] = {} @@ -321,28 +332,62 @@ def _is_containment_equivalent(first: _ScoredCandidate, second: _ScoredCandidate def _soft_prune(root: etree._Element) -> None: """在候选副本中删除确定的导航/表单和高噪声 token 子树。""" + metrics_by_element = _collect_soft_prune_metrics(root) for element in list(root.iterdescendants()): if not isinstance(element.tag, str): continue name = local_name(element) tokens = _tokens(element) - valuable = any( - isinstance(child.tag, str) and local_name(child) in _SEMANTIC_TAGS for child in element.iterdescendants() - ) - text = _normalized_text(" ".join(element.itertext())) - links = " ".join( - " ".join(link.itertext()) - for link in element.iterdescendants() - if isinstance(link.tag, str) and local_name(link) == "a" - ) - link_density = len(_normalized_text(links)) / max(1, len(text)) + metrics = metrics_by_element[element] + valuable = metrics.semantic_count > (1 if name in _SEMANTIC_TAGS else 0) + text_chars = metrics.text_chars + max(0, metrics.text_segments - 1) + link_chars = metrics.link_chars + max(0, metrics.link_count - 1) + link_density = link_chars / max(1, text_chars) should_remove = name in _BOILERPLATE_TAGS or ( - bool(tokens & _NEGATIVE_TOKENS) and not valuable and (len(text) < 200 or link_density > 0.5) + bool(tokens & _NEGATIVE_TOKENS) and not valuable and (text_chars < 200 or link_density > 0.5) ) if should_remove: _drop_tree_preserve_tail(element) +def _collect_soft_prune_metrics(root: etree._Element) -> dict[etree._Element, _SoftPruneMetrics]: + """单次后序遍历预计算每个元素的完整子树文本、链接和语义对象统计。""" + elements = [element for element in root.iter() if isinstance(element.tag, str)] + output: dict[etree._Element, _SoftPruneMetrics] = {} + for element in reversed(elements): + own_text = _normalized_text(element.text) + text_chars = len(own_text) + text_segments = 1 if own_text else 0 + link_chars = 0 + link_count = 0 + semantic_count = 1 if local_name(element) in _SEMANTIC_TAGS else 0 + for child in element: + if isinstance(child.tag, str): + child_metrics = output[child] + text_chars += child_metrics.text_chars + text_segments += child_metrics.text_segments + link_chars += child_metrics.link_chars + link_count += child_metrics.link_count + semantic_count += child_metrics.semantic_count + if local_name(child) == "a": + child_text_chars = child_metrics.text_chars + max(0, child_metrics.text_segments - 1) + if child_text_chars: + link_chars += child_text_chars + link_count += 1 + tail = _normalized_text(child.tail) + if tail: + text_chars += len(tail) + text_segments += 1 + output[element] = _SoftPruneMetrics( + text_chars=text_chars, + text_segments=text_segments, + link_chars=link_chars, + link_count=link_count, + semantic_count=semantic_count, + ) + return output + + def _drop_tree_preserve_tail(element: etree._Element) -> None: """删除噪声子树,同时把 tail 归还到相邻文本位置。""" parent = element.getparent() diff --git a/tests/unittest/test_flash_html.py b/tests/unittest/test_flash_html.py index dcaebd81..57d4b3d2 100644 --- a/tests/unittest/test_flash_html.py +++ b/tests/unittest/test_flash_html.py @@ -500,6 +500,28 @@ def test_html_auto_selection_precomputes_nested_subtree_penalties(monkeypatch: p assert normalization_calls < leaf_count * 8 +def test_html_soft_prune_precomputes_deep_subtree_text(monkeypatch: pytest.MonkeyPatch) -> None: + """验证 soft prune 只线性扫描深层候选中的大段文本。""" + depth = 200 + text = "x" * 4096 + root = lxml_html.fromstring("
" + "
" * depth + f"

{text}

" + "
" * depth + "
") + original_normalized_text = html_selector_module._normalized_text + normalized_input_chars = 0 + + def counted_normalized_text(value: str | None) -> str: + """累计送入文本规范化函数的原始字符数。""" + nonlocal normalized_input_chars + normalized_input_chars += len(value or "") + return original_normalized_text(value) + + monkeypatch.setattr(html_selector_module, "_normalized_text", counted_normalized_text) + + html_selector_module._soft_prune(root) + + assert normalized_input_chars <= len(text) * 2 + assert text in "".join(root.itertext()) + + def test_html_referenced_external_footnote_keeps_anchor_and_content() -> None: """验证正文候选外但被引用的 HTML footnote 会追加并生成可兑现 anchor。""" payload = b"""

Notes

@@ -1277,6 +1299,16 @@ def test_html_comments_count_toward_dom_node_budget(monkeypatch: pytest.MonkeyPa doc_analyze(b"", file_suffix="html") +@pytest.mark.parametrize("container", ["template", "form"]) +def test_html_ignores_stylesheets_beneath_discarded_active_subtrees(container: str) -> None: + """验证待删除活动子树内的 stylesheet 不会污染正文样式。""" + payload = f"<{container}>

Visible

".encode() + + markdown = render_markdown(doc_analyze(payload, file_suffix="html")[0]) + + assert "Visible" in markdown + + @pytest.mark.parametrize( ("styles", "single_limit", "total_limit", "expected_limit"), [