diff --git a/mineru/model/flash/_shared/markup/projector.py b/mineru/model/flash/_shared/markup/projector.py
index 2da129a8..723c9d59 100644
--- a/mineru/model/flash/_shared/markup/projector.py
+++ b/mineru/model/flash/_shared/markup/projector.py
@@ -6,7 +6,7 @@ from __future__ import annotations
import html
import re
from dataclasses import dataclass
-from typing import Protocol
+from typing import Protocol, TypeAlias
from lxml import etree # type: ignore[reportMissingImports]
@@ -98,6 +98,8 @@ _FOOTNOTE_TOKENS = frozenset(
}
)
_VISUAL_ELEMENT_TAGS = frozenset({"img", "image", "pre", "svg", "table"})
+_LIST_PAGE_BLOCK_TAGS = frozenset({"figure", "image", "img", "pre", "svg", "table"})
+_InlineProjectionSegment: TypeAlias = str | dict[str, object]
def local_name(element: etree._Element) -> str:
@@ -130,6 +132,17 @@ def _raw_visual_type(value: object) -> BlockType | None:
return None
+def _append_inline_segment(
+ segments: list[_InlineProjectionSegment],
+ segment: _InlineProjectionSegment,
+) -> None:
+ """追加行内投影片段,并合并相邻文本以保持稳定 block 粒度。"""
+ if isinstance(segment, str) and segments and isinstance(segments[-1], str):
+ segments[-1] += segment
+ elif not isinstance(segment, str) or segment:
+ segments.append(segment)
+
+
def entity_text(element: etree._Element) -> str:
"""把 lxml 保留的安全命名实体恢复为可见文本。"""
name = getattr(element, "name", "")
@@ -338,9 +351,18 @@ class MarkupProjector:
visibility_hidden: bool,
) -> list[dict[str, object]]:
"""转换标题或段落,并旁路其中的视觉 blocks。"""
- content, extras = self._render_inline_children(element, style, visibility_hidden)
blocks: list[dict[str, object]] = []
- if content.strip():
+ text_emitted = False
+ for segment in self._render_inline_children_ordered(element, style, visibility_hidden):
+ if not isinstance(segment, str):
+ blocks.append(segment)
+ continue
+ content = segment.strip()
+ if not content:
+ continue
+ if text_emitted:
+ blocks.append({"type": BlockType.TEXT, "content": content})
+ continue
if name == "h1" and (not self.single_document_title or not self.document_title_emitted):
block: dict[str, object] = {"type": BlockType.DOC_TITLE, "level": 1, "content": content.strip()}
self.document_title_emitted = True
@@ -357,7 +379,8 @@ class MarkupProjector:
if name.startswith("h") and (anchor := self.context.heading_anchor(element)):
block["anchor"] = anchor
blocks.append(block)
- return [*blocks, *extras]
+ text_emitted = True
+ return blocks
def _parse_note_element(
self,
@@ -385,20 +408,33 @@ class MarkupProjector:
visibility_hidden: bool = False,
) -> tuple[str, list[dict[str, object]]]:
"""渲染元素的连续行内内容,并旁路其中的视觉 blocks。"""
- parts = [] if visibility_hidden else [self._render_text(element.text, style)]
- extras: list[dict[str, object]] = []
+ segments = self._render_inline_children_ordered(element, style, visibility_hidden)
+ return (
+ "".join(segment for segment in segments if isinstance(segment, str)),
+ [segment for segment in segments if not isinstance(segment, str)],
+ )
+
+ def _render_inline_children_ordered(
+ self,
+ element: etree._Element,
+ style: TextStyle,
+ visibility_hidden: bool = False,
+ ) -> list[_InlineProjectionSegment]:
+ """按 DOM 顺序返回连续文本与旁路 block,保留 inline visual 前后边界。"""
+ segments: list[_InlineProjectionSegment] = []
+ if not visibility_hidden:
+ _append_inline_segment(segments, self._render_text(element.text, style))
for child in element:
if not isinstance(child.tag, str):
if not visibility_hidden:
- parts.append(self._render_text(entity_text(child), style))
- parts.append(self._render_text(child.tail, style))
+ _append_inline_segment(segments, self._render_text(entity_text(child), style))
+ _append_inline_segment(segments, self._render_text(child.tail, style))
continue
- rendered, child_extras = self._render_inline_element(child, style, visibility_hidden)
- parts.append(rendered)
- extras.extend(child_extras)
+ for segment in self._render_inline_element_ordered(child, style, visibility_hidden):
+ _append_inline_segment(segments, segment)
if not visibility_hidden:
- parts.append(self._render_text(child.tail, style))
- return "".join(parts), extras
+ _append_inline_segment(segments, self._render_text(child.tail, style))
+ return segments
def _render_inline_element(
self,
@@ -407,36 +443,54 @@ class MarkupProjector:
inherited_visibility_hidden: bool = False,
) -> tuple[str, list[dict[str, object]]]:
"""把一个行内元素转换为内部富文本协议和可选视觉块。"""
+ segments = self._render_inline_element_ordered(element, inherited, inherited_visibility_hidden)
+ return (
+ "".join(segment for segment in segments if isinstance(segment, str)),
+ [segment for segment in segments if not isinstance(segment, str)],
+ )
+
+ def _render_inline_element_ordered(
+ self,
+ element: etree._Element,
+ inherited: TextStyle,
+ inherited_visibility_hidden: bool = False,
+ ) -> list[_InlineProjectionSegment]:
+ """递归投影单个行内元素,并在嵌套 visual 位置保留顺序分段。"""
resolved = self.stylesheet.resolve(element, inherited, inherited_visibility_hidden)
if resolved.subtree_hidden:
- return "", []
+ return []
name = local_name(element)
if name in SKIPPED_TAGS:
- return "", []
+ return []
if name == "br":
- return ("", []) if resolved.visibility_hidden else ("\n", [])
+ return [] if resolved.visibility_hidden else ["\n"]
if name in {"img", "image"}:
- return ("", []) if resolved.visibility_hidden else ("", self._image_blocks(element))
+ return [] if resolved.visibility_hidden else self._image_blocks(element)
if name == "math":
if resolved.visibility_hidden:
- return "", []
+ return []
formula = self._formula_extraction(element)
if formula is not None:
- return f"{html.escape(code, quote=False)}" if code else ""), []
- content, extras = self._render_inline_children(element, resolved.text, resolved.visibility_hidden)
+ return [f"{html.escape(code, quote=False)}"] if code else []
+ if name in BLOCK_TAGS:
+ return self._parse_block(element, inherited, inherited_visibility_hidden)
+ segments = self._render_inline_children_ordered(element, resolved.text, resolved.visibility_hidden)
if name == "a":
href = element.get("href") or element.get(_XLINK_HREF) or ""
target = self.context.resolve_link(href)
- if target and content.strip():
- return render_inline_hyperlink(content, target), extras
- return content, extras
+ if target:
+ return [
+ render_inline_hyperlink(segment, target) if isinstance(segment, str) and segment.strip() else segment
+ for segment in segments
+ ]
+ return segments
@staticmethod
def _render_text(value: str | None, style: TextStyle) -> str:
@@ -506,26 +560,18 @@ class MarkupProjector:
]
annotation_elements = {child for child, _ in annotations}
mineru_figure = "mineru-figure" in (element.get("class") or "").casefold().split()
- blocks: list[dict[str, object]] = []
- for child in element:
- if not isinstance(child.tag, str) or child in annotation_elements:
- continue
- name = local_name(child)
- if name in {"img", "image"}:
- child_style = self.stylesheet.resolve(child, style, visibility_hidden)
- if not child_style.subtree_hidden and not child_style.visibility_hidden:
- blocks.extend(self._image_blocks(child, emit_alt_caption=not mineru_figure and not annotations))
- else:
- child_blocks = (
- self._parse_block(child, style, visibility_hidden)
- if name in BLOCK_TAGS
- else self._render_inline_element(child, style, visibility_hidden)[1]
- )
- blocks.extend(child_blocks)
+ blocks = self._parse_figure_contents(
+ element,
+ style,
+ visibility_hidden,
+ annotation_elements=annotation_elements,
+ emit_alt_caption=not mineru_figure and not annotations,
+ )
visual_types = {
_raw_visual_type(block.get("type")) for block in blocks if _raw_visual_type(block.get("type")) is not None
}
+ visual_type = next(iter(visual_types)) if len(visual_types) == 1 else None
annotation_blocks: list[dict[str, object]] = []
for annotation, kind in annotations:
resolved = self.stylesheet.resolve(annotation, style, visibility_hidden)
@@ -535,9 +581,7 @@ class MarkupProjector:
annotation_blocks.extend(extras)
if not content.strip():
continue
- if len(visual_types) == 1:
- visual_type = next(iter(visual_types))
- assert visual_type is not None
+ if visual_type is not None:
annotation_blocks.append(
{
"type": VISUAL_TYPE_MAPPING[visual_type][kind],
@@ -546,7 +590,66 @@ class MarkupProjector:
)
else:
annotation_blocks.append({"type": BlockType.TEXT, "content": content.strip()})
- blocks.extend(annotation_blocks)
+ if visual_type is None or not annotation_blocks:
+ blocks.extend(annotation_blocks)
+ return blocks
+
+ visual_positions = [index for index, block in enumerate(blocks) if _raw_visual_type(block.get("type")) == visual_type]
+ assert visual_positions
+ insert_at = visual_positions[-1] + 1
+ blocks[insert_at:insert_at] = annotation_blocks
+ return blocks
+
+ def _parse_figure_contents(
+ self,
+ element: etree._Element,
+ style: TextStyle,
+ visibility_hidden: bool,
+ *,
+ annotation_elements: set[etree._Element],
+ emit_alt_caption: bool,
+ ) -> list[dict[str, object]]:
+ """按 DOM 顺序缓冲 figure 文本,并在 visual extras 前后切分正文 block。"""
+ blocks: list[dict[str, object]] = []
+ inline_parts: list[str] = [] if visibility_hidden else [self._render_text(element.text, style)]
+
+ def flush_inline() -> None:
+ """把 figure 当前连续文本写为普通正文 block。"""
+ content = "".join(inline_parts).strip()
+ inline_parts.clear()
+ if content:
+ blocks.append({"type": BlockType.TEXT, "content": content})
+
+ for child in element:
+ if not isinstance(child.tag, str):
+ if not visibility_hidden:
+ inline_parts.append(self._render_text(entity_text(child), style))
+ inline_parts.append(self._render_text(child.tail, style))
+ continue
+ if child in annotation_elements:
+ if not visibility_hidden:
+ inline_parts.append(self._render_text(child.tail, style))
+ continue
+
+ name = local_name(child)
+ if name in {"img", "image"}:
+ flush_inline()
+ child_style = self.stylesheet.resolve(child, style, visibility_hidden)
+ if not child_style.subtree_hidden and not child_style.visibility_hidden:
+ blocks.extend(self._image_blocks(child, emit_alt_caption=emit_alt_caption))
+ elif name in BLOCK_TAGS:
+ flush_inline()
+ blocks.extend(self._parse_block(child, style, visibility_hidden))
+ else:
+ for segment in self._render_inline_element_ordered(child, style, visibility_hidden):
+ if isinstance(segment, str):
+ inline_parts.append(segment)
+ else:
+ flush_inline()
+ blocks.append(segment)
+ if not visibility_hidden:
+ inline_parts.append(self._render_text(child.tail, style))
+ flush_inline()
return blocks
def _has_contextual_visual_annotation(self, element: etree._Element) -> bool:
@@ -810,6 +913,8 @@ class MarkupProjector:
visibility_hidden: bool = False,
) -> tuple[dict[str, object] | None, list[dict[str, object]]]:
"""解析有序/无序列表,并投影为连续阿拉伯编号结构。"""
+ if self._list_contains_page_blocks(element):
+ return self._parse_list_with_page_blocks(element, style, visibility_hidden)
ordered = local_name(element) == "ol"
items = [child for child in element if isinstance(child.tag, str) and local_name(child) == "li"]
if not items:
@@ -876,6 +981,128 @@ class MarkupProjector:
block["start"] = self._ordered_list_start(element)
return block, extras
+ @staticmethod
+ def _list_contains_page_blocks(element: etree._Element) -> bool:
+ """判断列表是否含必须提升为页面兄弟的 visual/code 子树。"""
+ return any(
+ isinstance(candidate.tag, str) and local_name(candidate) in _LIST_PAGE_BLOCK_TAGS
+ for candidate in element.iterdescendants()
+ )
+
+ def _parse_list_with_page_blocks(
+ self,
+ element: etree._Element,
+ style: TextStyle,
+ visibility_hidden: bool,
+ ) -> tuple[dict[str, object] | None, list[dict[str, object]]]:
+ """把含 visual 的列表切成有序 list/text/page block 片段,保持 DOM 阅读顺序。"""
+ ordered = local_name(element) == "ol"
+ list_start = self._ordered_list_start(element) if ordered else 1
+ items = [child for child in element if isinstance(child.tag, str) and local_name(child) == "li"]
+ pending_children: list[dict[str, object]] = []
+ pending_start = list_start
+ output: list[dict[str, object]] = []
+ visible_item_ordinal = 0
+ has_page_blocks = False
+
+ def flush_pending() -> None:
+ """把当前连续列表项写为一个顶层 list block。"""
+ nonlocal pending_children
+ if not pending_children:
+ return
+ output.append(self._build_raw_list_block(pending_children, ordered=ordered, start=pending_start))
+ pending_children = []
+
+ for item in items:
+ item_style = self.stylesheet.resolve(item, style, visibility_hidden)
+ if item_style.subtree_hidden:
+ continue
+ if self.context.note_anchor(item) is not None:
+ flush_pending()
+ output.extend(self._parse_note_element(item, item_style.text, item_style.visibility_hidden))
+ has_page_blocks = True
+ continue
+
+ segments = self._normalize_list_item_segments(
+ self._render_inline_children_ordered(item, item_style.text, item_style.visibility_hidden)
+ )
+ page_positions = [
+ index
+ for index, segment in enumerate(segments)
+ if not isinstance(segment, str) and segment.get("type") != BlockType.LIST
+ ]
+ if not page_positions:
+ item_children = self._list_item_children(segments)
+ if item_children:
+ if not pending_children:
+ pending_start = list_start + visible_item_ordinal
+ pending_children.extend(item_children)
+ visible_item_ordinal += 1
+ continue
+
+ first_page_position = page_positions[0]
+ prefix_children = self._list_item_children(segments[:first_page_position])
+ if not pending_children:
+ pending_start = list_start + visible_item_ordinal
+ pending_children.extend(prefix_children or [{"type": BlockType.TEXT, "content": ""}])
+ flush_pending()
+
+ for segment in segments[first_page_position:]:
+ if isinstance(segment, str):
+ content = segment.strip()
+ if content:
+ output.append({"type": BlockType.TEXT, "content": content})
+ else:
+ output.append(segment)
+ visible_item_ordinal += 1
+ has_page_blocks = True
+
+ if not has_page_blocks:
+ return (
+ self._build_raw_list_block(pending_children, ordered=ordered, start=list_start) if pending_children else None,
+ [],
+ )
+ flush_pending()
+ return None, output
+
+ @staticmethod
+ def _normalize_list_item_segments(
+ segments: list[_InlineProjectionSegment],
+ ) -> list[_InlineProjectionSegment]:
+ """把列表内部普通 text block 还原为文本片段,保留 visual/list 页面边界。"""
+ normalized: list[_InlineProjectionSegment] = []
+ for segment in segments:
+ if isinstance(segment, dict) and segment.get("type") == BlockType.TEXT:
+ _append_inline_segment(normalized, str(segment.get("content") or ""))
+ else:
+ _append_inline_segment(normalized, segment)
+ return normalized
+
+ @staticmethod
+ def _list_item_children(segments: list[_InlineProjectionSegment]) -> list[dict[str, object]]:
+ """把无页面 visual 的列表片段收敛为一个文本叶子及其嵌套列表。"""
+ content = "".join(segment for segment in segments if isinstance(segment, str)).strip()
+ children = [{"type": BlockType.TEXT, "content": content}] if content else []
+ children.extend(segment for segment in segments if isinstance(segment, dict) and segment.get("type") == BlockType.LIST)
+ return children
+
+ @staticmethod
+ def _build_raw_list_block(
+ children: list[dict[str, object]],
+ *,
+ ordered: bool,
+ start: int,
+ ) -> dict[str, object]:
+ """构造一段可由既有无坐标后处理编号的 raw list block。"""
+ block: dict[str, object] = {
+ "type": BlockType.LIST,
+ "attribute": "ordered" if ordered else "unordered",
+ "content": children,
+ }
+ if ordered:
+ block["start"] = start
+ return block
+
@staticmethod
def _ordered_list_start(element: etree._Element) -> int:
"""读取有序列表唯一通用起始值,非法或负值统一回退为一。"""
diff --git a/mineru/render/_internal/docx/renderer.py b/mineru/render/_internal/docx/renderer.py
index a219ad90..f3ee477e 100644
--- a/mineru/render/_internal/docx/renderer.py
+++ b/mineru/render/_internal/docx/renderer.py
@@ -114,6 +114,11 @@ _ANNOTATION_FOOTNOTE_TYPES = {
}
+def _has_block_image_payload(block: ImagePayloadBlock) -> bool:
+ """判断统一图片载荷是否包含 sidecar、data URI 或远程 URL。"""
+ return block.image_path is not None or block.image_base64 is not None or block.image_url is not None
+
+
class _DocxRenderer:
"""持有单次 DOCX 渲染所需的 document、素材解析器和书签状态。"""
@@ -230,7 +235,7 @@ class _DocxRenderer:
except DocxFormulaError as exc:
logger.warning("DOCX display formula fallback: {} ({})", exc, context.location())
- if block.image_base64 is not None or block.image_path is not None:
+ if _has_block_image_payload(block):
try:
self._append_block_image(block, context=context, alt_text="formula")
return
@@ -350,7 +355,7 @@ class _DocxRenderer:
self._append_html_tables(child.content, context=context, depth=0)
except DocxRenderError as exc:
logger.warning("DOCX HTML table fallback: {}", exc)
- if child.image_base64 is None and child.image_path is None:
+ if not _has_block_image_payload(child):
raise self._render_error(
"HTML table cannot be materialized and has no image fallback",
context,
@@ -360,7 +365,7 @@ class _DocxRenderer:
if child.content and not child.content.isspace():
paragraph = self.document.add_paragraph(style=SPATIAL_TABLE_STYLE)
paragraph.add_run(sanitize_xml_text(child.content, context=context))
- elif child.image_base64 is not None or child.image_path is not None:
+ elif _has_block_image_payload(child):
self._append_block_image(child, context=context, alt_text="table")
else:
raise self._render_error(
@@ -376,7 +381,7 @@ class _DocxRenderer:
"""先写图表图片,再把 HTML 结构化数据追加为原生表格。"""
for child in block.content:
if isinstance(child, ChartBodyBlock):
- has_image = child.image_base64 is not None or child.image_path is not None
+ has_image = _has_block_image_payload(child)
if has_image:
self._append_block_image(
child,
diff --git a/tests/unittest/test_docx_render.py b/tests/unittest/test_docx_render.py
index e72a1c9b..d613d245 100644
--- a/tests/unittest/test_docx_render.py
+++ b/tests/unittest/test_docx_render.py
@@ -757,6 +757,60 @@ def test_spatial_table_without_text_uses_sidecar_resolver() -> None:
assert len(document.inline_shapes) == 1
+def test_remote_only_equation_table_and_chart_use_docx_link_fallbacks() -> None:
+ """验证所有远程-only 图片载荷都会进入 DOCX 可点击链接回退。"""
+ middle = _middle(
+ _page(
+ 0,
+ EquationBlock(
+ type="equation",
+ index=0,
+ content="",
+ image_url="https://example.com/formula.png",
+ ),
+ TableBlock(
+ type="table",
+ index=1,
+ content=[
+ TableBodyBlock(
+ type="table_body",
+ index=1,
+ content="",
+ image_url="https://example.com/table.png",
+ )
+ ],
+ ),
+ ChartBlock(
+ type="chart",
+ index=2,
+ sub_type="bar",
+ content=[
+ ChartBodyBlock(
+ type="chart_body",
+ index=2,
+ content="",
+ image_url="https://example.com/chart.png",
+ )
+ ],
+ ),
+ )
+ )
+
+ result = render_docx(middle)
+ document = Document(BytesIO(result))
+ relationships = _part(result, "word/_rels/document.xml.rels")
+
+ assert [paragraph.text for paragraph in document.paragraphs] == ["formula", "table", "bar"]
+ assert all(
+ target in relationships
+ for target in (
+ "https://example.com/formula.png",
+ "https://example.com/table.png",
+ "https://example.com/chart.png",
+ )
+ )
+
+
@pytest.mark.parametrize("content", ["", " \n\t"])
def test_spatial_table_without_text_or_image_raises_contextual_error(content: str) -> None:
"""验证空间表格既无有效文本也无图片时抛出带父表格定位的异常。"""
diff --git a/tests/unittest/test_flash_epub.py b/tests/unittest/test_flash_epub.py
index 9e58fdb9..f3fb14dc 100644
--- a/tests/unittest/test_flash_epub.py
+++ b/tests/unittest/test_flash_epub.py
@@ -364,6 +364,38 @@ def test_epub_figure_skips_hidden_direct_images(attribute: str, value: str) -> N
package.close()
+def test_epub_figure_preserves_direct_text_and_child_tails() -> None:
+ """验证共享 projector 不丢弃 XHTML figure 的直属文本及图片、caption tail。"""
+ package = EpubPackage(build_epub_fixture())
+ chapter_path = "EPUB/text/ch1.xhtml"
+ try:
+ root = package.xml_part(chapter_path, allow_external_doctype=True)
+ figure = next(element for element in root.iter() if isinstance(element.tag, str) and element.tag.endswith("}figure"))
+ image = next(child for child in figure if isinstance(child.tag, str) and child.tag.endswith("}img"))
+ caption = next(child for child in figure if isinstance(child.tag, str) and child.tag.endswith("}figcaption"))
+ figure.text = "Before"
+ image.tail = "After"
+ caption.tail = "Tail"
+
+ anchors = build_anchor_registry([(chapter_path, root)], package)
+ blocks = EpubChapterConverter(package, chapter_path, root, anchors).convert()
+ relevant = [
+ block
+ for block in blocks
+ if block.get("type") in {BlockType.IMAGE, BlockType.IMAGE_CAPTION}
+ or (isinstance(block.get("content"), str) and block.get("content") in {"Before", "AfterTail"})
+ ]
+
+ assert [(block["type"], block.get("content")) for block in relevant] == [
+ (BlockType.TEXT, "Before"),
+ (BlockType.IMAGE, ""),
+ (BlockType.IMAGE_CAPTION, "Dot caption"),
+ (BlockType.TEXT, "AfterTail"),
+ ]
+ finally:
+ package.close()
+
+
def test_public_parser_rejects_epub_page_range(tmp_path: Path) -> None:
"""验证 EPUB 公共 Parser 只接受整本解析。"""
source = tmp_path / "book.epub"
diff --git a/tests/unittest/test_flash_html.py b/tests/unittest/test_flash_html.py
index bb72e2bf..6ded7dff 100644
--- a/tests/unittest/test_flash_html.py
+++ b/tests/unittest/test_flash_html.py
@@ -387,6 +387,64 @@ def test_html_mineru_table_figure_rebinds_renderer_caption() -> None:
assert markdown.count("Visible table caption") == 1
+def test_html_figure_preserves_direct_and_inline_text_around_visuals() -> None:
+ """验证 figure 的直属文本、行内容器和 child tail 按 visual 前后顺序进入 raw blocks。"""
+ payload = b"""
After
+ Before
After
After