fix: harden native parser output boundaries

This commit is contained in:
myhloli
2026-08-27 01:28:15 +08:00
parent fc53c49b50
commit 97d00a3bbd
9 changed files with 290 additions and 23 deletions
+66 -4
View File
@@ -20,6 +20,7 @@ MAX_CSV_ROWS: Final = 1_048_576
MAX_CSV_COLUMNS: Final = 16_384
# CSV 会把每个槽位实体化为 HTML/DOM 节点,预算需显著低于稀疏电子表格投影上限。
MAX_CSV_GRID_SLOTS: Final = 250_000
MAX_CSV_RENDERED_BYTES: Final = 256 * 1024 * 1024
_DELIMITER_CANDIDATES: Final = (",", ";", "\t", "|")
_DELIMITER_SAMPLE_RECORDS: Final = 20
@@ -255,17 +256,78 @@ def _render_field_html(value: str) -> str:
return html.escape(normalized, quote=True).replace("\n", "<br>")
def _rendered_field_utf8_bytes(value: str, remaining_budget: int) -> int:
"""在不创建转义字符串的前提下计算字段渲染后的 UTF-8 字节数。"""
rendered_bytes = 0
index = 0
while index < len(value):
char = value[index]
codepoint = ord(char)
if char == "\r":
if index + 1 < len(value) and value[index + 1] == "\n":
index += 1
addition = len("<br>")
elif char == "\n":
addition = len("<br>")
elif codepoint <= 0x08 or codepoint in {0x0B, 0x0C, 0x7F} or 0x0E <= codepoint <= 0x1F:
addition = 3
elif 0xD800 <= codepoint <= 0xDFFF:
addition = 3
elif char == "&":
addition = len("&amp;")
elif char in {"<", ">"}:
addition = len("&lt;")
elif char in {'"', "'"}:
addition = len("&quot;")
elif codepoint <= 0x7F:
addition = 1
elif codepoint <= 0x7FF:
addition = 2
elif codepoint <= 0xFFFF:
addition = 3
else:
addition = 4
rendered_bytes += addition
if rendered_bytes > remaining_budget:
raise ValueError(f"CSV exceeds max_rendered_bytes={MAX_CSV_RENDERED_BYTES}")
index += 1
return rendered_bytes
def _charge_rendered_bytes(used_bytes: int, additional_bytes: int) -> int:
"""累计 CSV HTML 输出预算,并在写入 StringIO 前拒绝超限内容。"""
if additional_bytes < 0 or used_bytes > MAX_CSV_RENDERED_BYTES - additional_bytes:
raise ValueError(f"CSV exceeds max_rendered_bytes={MAX_CSV_RENDERED_BYTES}")
return used_bytes + additional_bytes
def _rows_to_html(rows: list[list[str]], *, has_header: bool) -> str:
"""增量构造安全表格 HTML,避免为每个单元格保留独立字符串对象。"""
output = StringIO()
rendered_bytes = 0
rendered_bytes = _charge_rendered_bytes(rendered_bytes, len("<table>"))
output.write("<table>")
for row_index, row in enumerate(rows):
tag = "th" if has_header and row_index == 0 else "td"
output.write("\n <tr>")
row_prefix = "\n <tr>"
rendered_bytes = _charge_rendered_bytes(rendered_bytes, len(row_prefix))
output.write(row_prefix)
for value in row:
output.write(f"\n <{tag}>{_render_field_html(value)}</{tag}>")
output.write("\n </tr>")
output.write("\n</table>")
cell_prefix = f"\n <{tag}>"
cell_suffix = f"</{tag}>"
rendered_bytes = _charge_rendered_bytes(rendered_bytes, len(cell_prefix) + len(cell_suffix))
remaining_budget = MAX_CSV_RENDERED_BYTES - rendered_bytes
field_bytes = _rendered_field_utf8_bytes(value, remaining_budget)
rendered_bytes = _charge_rendered_bytes(rendered_bytes, field_bytes)
output.write(cell_prefix)
output.write(_render_field_html(value))
output.write(cell_suffix)
row_suffix = "\n </tr>"
rendered_bytes = _charge_rendered_bytes(rendered_bytes, len(row_suffix))
output.write(row_suffix)
table_suffix = "\n</table>"
_charge_rendered_bytes(rendered_bytes, len(table_suffix))
output.write(table_suffix)
return output.getvalue()
+38 -13
View File
@@ -489,7 +489,7 @@ class EpubChapterConverter:
if name == "figure":
return self._parse_figure(element, resolved.text)
if name == "svg":
return self._parse_svg(element)
return self._parse_svg(element, resolved.text)
return self._parse_container_contents(element, resolved.text)
def _parse_note_element(self, element: etree._Element, style: TextStyle) -> list[dict[str, object]]:
@@ -622,20 +622,44 @@ class EpubChapterConverter:
blocks.append({"type": BlockType.TEXT, "content": html.escape(caption, quote=False)})
return blocks
def _parse_svg(self, element: etree._Element) -> list[dict[str, object]]:
def _visible_svg_text(self, element: etree._Element, style: TextStyle) -> str:
"""递归提取 SVG 可见文本,并排除隐藏后代的内容。"""
parts = [_clean_text_node(element.text)]
for child in element:
if not isinstance(child.tag, str):
parts.append(_clean_text_node(_entity_text(child)))
parts.append(_clean_text_node(child.tail))
continue
resolved = self.stylesheet.resolve(child, style)
if not resolved.hidden:
parts.append(self._visible_svg_text(child, resolved.text))
parts.append(_clean_text_node(child.tail))
return _WHITESPACE_RE.sub(" ", html.unescape("".join(parts))).strip()
def _parse_svg(self, element: etree._Element, style: TextStyle) -> list[dict[str, object]]:
"""从 SVG 尽力提取 title/desc/text 和包内栅格 image。"""
blocks: list[dict[str, object]] = []
texts: list[str] = []
for child in element.iter():
if not isinstance(child.tag, str):
continue
name = _local_name(child)
if name in {"title", "desc", "text"}:
value = _visible_text(child)
if value and value not in texts:
texts.append(value)
elif name == "image":
blocks.extend(self._image_blocks(child))
def visit(parent: etree._Element, inherited: TextStyle) -> None:
"""按 SVG 树顺序访问可见候选节点,并让祖先隐藏状态截断子树。"""
for child in parent:
if not isinstance(child.tag, str):
continue
resolved = self.stylesheet.resolve(child, inherited)
if resolved.hidden:
continue
name = _local_name(child)
if name in {"title", "desc", "text"}:
value = self._visible_svg_text(child, resolved.text)
if value and value not in texts:
texts.append(value)
elif name == "image":
blocks.extend(self._image_blocks(child))
else:
visit(child, resolved.text)
visit(element, style)
if texts:
blocks.insert(0, {"type": BlockType.TEXT, "content": html.escape("\n".join(texts), quote=False)})
return blocks
@@ -880,7 +904,8 @@ def convert_svg_spine(
"""把 standalone SVG spine item 尽力转换为文本和包内栅格图片。"""
empty_registry = EpubAnchorRegistry([], package)
converter = EpubChapterConverter(package, chapter_path, root, empty_registry)
return converter._parse_svg(root)
resolved = converter.stylesheet.resolve(root, TextStyle())
return [] if resolved.hidden else converter._parse_svg(root, resolved.text)
__all__ = [
+4 -1
View File
@@ -3,6 +3,7 @@
from __future__ import annotations
import html
import re
from dataclasses import dataclass
from typing import Any, BinaryIO, Iterator
@@ -251,6 +252,8 @@ def _parse_odp_pages(context: _OdfContext) -> list[list[dict[str, Any]]]:
for page in context.body:
if page.tag != qname("draw", "page"):
continue
if not context.styles.drawing_page_is_visible(page):
continue
positioned: list[_PositionedBlocks] = []
for order, (shape, x, y) in enumerate(_iter_slide_shapes(page)):
presentation_class = shape.get(qname("presentation", "class"), "")
@@ -324,7 +327,7 @@ def _parse_ods_pages(context: _OdfContext) -> list[list[dict[str, Any]]]:
continue
if not context.styles.table_is_visible(sheet.get(qname("table", "style-name"))):
continue
name = sheet.get(qname("table", "name"), "Sheet")
name = html.escape(sheet.get(qname("table", "name"), "Sheet"), quote=False)
sheet_pages.append((name, _sheet_blocks(sheet, parser)))
if sum(bool(blocks) for _, blocks in sheet_pages) > 1:
for name, blocks in sheet_pages:
+3 -1
View File
@@ -73,7 +73,9 @@ def extract_odf_metadata(file_binary: BinaryIO, suffix: OdfSuffix) -> dict[str,
if suffix == "odt":
page_count = _odt_page_count(meta_root)
elif suffix == "odp":
page_count = sum(1 for child in body if child.tag == qname("draw", "page"))
page_count = sum(
1 for child in body if child.tag == qname("draw", "page") and styles.drawing_page_is_visible(child)
)
else:
page_count = _visible_sheet_count(body, styles)
return {
+42
View File
@@ -23,6 +23,7 @@ class _StyleDefinition:
text_delta: TextStyleDelta
master_page_name: str | None
table_display: bool | None
drawing_page_visible: bool | None
class OdfStyles:
@@ -35,6 +36,7 @@ class OdfStyles:
self._list_styles: dict[str, dict[int, ListLevel]] = {}
self._resolved_text: dict[tuple[str, str], TextStyleDelta] = {}
self._resolved_table_display: dict[str, bool | None] = {}
self._resolved_drawing_page_visibility: dict[str, bool | None] = {}
self._master_pages: dict[str, etree._Element] = {}
for root in roots:
if root is not None:
@@ -59,6 +61,7 @@ class OdfStyles:
text_delta=self._text_delta(style),
master_page_name=style.get(qname("style", "master-page-name")),
table_display=self._table_display(style),
drawing_page_visible=self._drawing_page_visibility(style),
)
for list_style in root.iter(qname("text", "list-style")):
name = list_style.get(qname("style", "name"))
@@ -133,6 +136,17 @@ class OdfStyles:
return None
return display.casefold() != "false"
@staticmethod
def _drawing_page_visibility(style: etree._Element) -> bool | None:
"""读取 drawing-page 样式的 presentation visibility。"""
properties = style.find(qname("style", "drawing-page-properties"))
if properties is None:
return None
visibility = properties.get(qname("presentation", "visibility"))
if visibility is None:
return None
return visibility.casefold() != "hidden"
@staticmethod
def _parse_list_style(element: etree._Element) -> dict[int, ListLevel]:
"""解析列表样式的层级、类型和通用起始值。"""
@@ -258,6 +272,34 @@ class OdfStyles:
self._resolved_table_display[style_name] = resolved
return resolved is not False
def drawing_page_is_visible(self, page: etree._Element) -> bool:
"""解析 ODP 页面直接属性或 drawing-page 样式中的隐藏状态。"""
direct_visibility = page.get(qname("presentation", "visibility"))
if direct_visibility is not None:
return direct_visibility.casefold() != "hidden"
style_name = page.get(qname("draw", "style-name"))
if not style_name:
return True
if style_name in self._resolved_drawing_page_visibility:
return self._resolved_drawing_page_visibility[style_name] is not False
seen: set[str] = set()
current = style_name
resolved: bool | None = None
while current:
if current in seen:
logger.warning("ODF style inheritance cycle detected: family=drawing-page, style={}", current)
break
seen.add(current)
definition = self._styles.get(("drawing-page", current))
if definition is None:
break
if definition.drawing_page_visible is not None:
resolved = definition.drawing_page_visible
break
current = definition.parent or ""
self._resolved_drawing_page_visibility[style_name] = resolved
return resolved is not False
def master_page(self, name: str | None) -> etree._Element | None:
"""返回指定 master-page;空名称时优先使用第一个定义。"""
if name and name in self._master_pages:
+12 -3
View File
@@ -344,6 +344,15 @@ def _column_index(label: str) -> int | None:
return result - 1 if result <= MAX_GRID_SLOTS else None
def _row_index(label: str) -> int | None:
"""在整数转换前把 A1 地址中的行号约束到共享网格预算。"""
normalized = label.lstrip("0")
if not normalized or len(normalized) > len(str(MAX_GRID_SLOTS)):
return None
result = int(normalized)
return result - 1 if result <= MAX_GRID_SLOTS else None
def parse_cell_range_bounds(address: str) -> tuple[int, int, int, int] | None:
"""从 ODF cell-range-address 中提取零基闭区间边界。"""
matches = list(_CELL_ADDRESS_RE.finditer(address or ""))
@@ -351,11 +360,11 @@ def parse_cell_range_bounds(address: str) -> tuple[int, int, int, int] | None:
return None
first = matches[0]
last = matches[-1]
row_start = int(first.group("row")) - 1
row_start = _row_index(first.group("row"))
col_start = _column_index(first.group("col"))
row_end = int(last.group("row")) - 1
row_end = _row_index(last.group("row"))
col_end = _column_index(last.group("col"))
if col_start is None or col_end is None:
if row_start is None or col_start is None or row_end is None or col_end is None:
return None
return min(row_start, row_end), max(row_start, row_end), min(col_start, col_end), max(col_start, col_end)
+22
View File
@@ -146,6 +146,28 @@ def test_csv_grid_limit_short_circuits_before_trailing_malformed_record(monkeypa
CsvModel().predict(BytesIO(b'a,b\n1,2\n"unterminated'))
def test_csv_rendered_budget_fails_before_materializing_escaped_field(monkeypatch: pytest.MonkeyPatch) -> None:
"""验证 HTML 展开超限时不会先创建放大的字段字符串。"""
monkeypatch.setattr(csv_module, "MAX_CSV_RENDERED_BYTES", 64)
def unexpected_escape(_value: str) -> str:
"""输出预算应在进入 html.escape 前拒绝字段。"""
pytest.fail("oversized CSV field reached HTML escaping")
monkeypatch.setattr(csv_module, "_render_field_html", unexpected_escape)
with pytest.raises(ValueError, match="max_rendered_bytes"):
csv_module._rows_to_html([["&" * 20]], has_header=False)
def test_csv_rendered_size_estimator_matches_html_escape_semantics() -> None:
"""验证特殊字符、换行、控制符和非 ASCII 文本的 UTF-8 预算精确。"""
value = "&<>\"'\r\n\x01"
rendered = csv_module._render_field_html(value)
assert csv_module._rendered_field_utf8_bytes(value, 1_000) == len(rendered.encode())
def test_csv_default_grid_budget_rejects_wide_dom_before_rendering() -> None:
"""验证默认预算在宽空表生成数十万 HTML 节点前拒绝输入。"""
assert csv_module.MAX_CSV_GRID_SLOTS == 250_000
+30 -1
View File
@@ -22,7 +22,7 @@ from mineru.errors import InvalidRequestError
from mineru.model.flash import EpubModel
from mineru.model.flash.epub import EpubEncryptedError, EpubPackage, EpubParseError, EpubResourceLimitError, detect_epub
from mineru.model.flash.epub.styles import EpubStylesheet, TextStyle
from mineru.model.flash.epub.xhtml import EpubChapterConverter, build_anchor_registry
from mineru.model.flash.epub.xhtml import EpubChapterConverter, build_anchor_registry, convert_svg_spine
from mineru.parser import MinerUParser, parse, parse_async
from mineru.parser import api_server
from mineru.parser.api_server import CreateJobRequest, FileStore
@@ -502,6 +502,35 @@ def test_epub_corrupt_chapter_keeps_empty_spine_placeholder() -> None:
assert "SVG text" in middle.pages[2].blocks[0].content # type: ignore[union-attr]
def test_epub_svg_extraction_skips_hidden_descendants_and_hidden_root() -> None:
"""验证 standalone SVG 不提取隐藏文本、隐藏图片或隐藏祖先子树。"""
package = EpubPackage(build_epub_fixture())
path = "EPUB/fixed/page.svg"
try:
root = package.xml_part(path)
assert root is not None
namespace = etree.QName(root).namespace
original_text = next(root.iter(f"{{{namespace}}}text"))
original_text.set("style", "display: none")
original_image = next(root.iter(f"{{{namespace}}}image"))
original_image.set("style", "visibility: hidden")
hidden_group = etree.SubElement(root, f"{{{namespace}}}g", style="display: none")
etree.SubElement(hidden_group, f"{{{namespace}}}text").text = "hidden group text"
etree.SubElement(root, f"{{{namespace}}}text").text = "visible graphic label"
blocks = convert_svg_spine(package, path, root)
assert "visible graphic label" in str(blocks)
assert "SVG text" not in str(blocks)
assert "hidden group text" not in str(blocks)
assert all(block["type"] != BlockType.IMAGE for block in blocks)
root.set("style", "display: none")
assert convert_svg_spine(package, path, root) == []
finally:
package.close()
def test_epub_malformed_resource_and_link_references_degrade_locally() -> None:
"""验证非法 URI 只丢弃样式、图片或链接目标,不阻断章节正文。"""
package = EpubPackage(build_epub_fixture())
+73
View File
@@ -582,6 +582,15 @@ def test_odf_rejects_overlong_chart_columns_before_bigint_conversion(monkeypatch
assert odf_table_module.parse_cell_range_bounds(f"local-table.{'A' * 100_000}1") is None
def test_odf_rejects_overlong_chart_rows_before_bigint_conversion(monkeypatch: pytest.MonkeyPatch) -> None:
"""验证 chart A1 行号在 int 转换前受共享网格预算约束。"""
monkeypatch.setattr(odf_table_module, "MAX_GRID_SLOTS", 4)
assert odf_table_module.parse_cell_range_bounds("local-table.A4:B4") == (3, 3, 0, 1)
assert odf_table_module.parse_cell_range_bounds("local-table.A5:B5") is None
assert odf_table_module.parse_cell_range_bounds(f"local-table.A{'9' * 100_000}") is None
@pytest.mark.parametrize(
"target",
[
@@ -737,6 +746,70 @@ def test_odf_flattened_titles_and_notes_keep_literal_protocol_escaped() -> None:
assert "javascript:alert(1)" not in relationships
def test_ods_sheet_titles_escape_literal_inline_protocol() -> None:
"""验证多 sheet 标题不会把名称中的内部协议重建为活动链接。"""
literal = "&lt;hyperlink&gt;&lt;text&gt;x&lt;/text&gt;&lt;url&gt;javascript:alert(1)&lt;/url&gt;&lt;/hyperlink&gt;"
content = f"""<office:document-content
xmlns:office="urn:oasis:names:tc:opendocument:xmlns:office:1.0"
xmlns:table="urn:oasis:names:tc:opendocument:xmlns:table:1.0"
xmlns:text="urn:oasis:names:tc:opendocument:xmlns:text:1.0">
<office:body><office:spreadsheet>
<table:table table:name="{literal}"><table:table-row><table:table-cell>
<text:p>A</text:p>
</table:table-cell></table:table-row></table:table>
<table:table table:name="Safe"><table:table-row><table:table-cell>
<text:p>B</text:p>
</table:table-cell></table:table-row></table:table>
</office:spreadsheet></office:body></office:document-content>"""
middle, model = doc_analyze(build_odf_package("ods", content), file_suffix="ods")
title = model.pages[0][0]["content"]
markdown = render_markdown(middle)
docx = render_docx(middle)
with ZipFile(BytesIO(docx)) as package:
relationships = package.read("word/_rels/document.xml.rels").decode("utf-8")
assert title.startswith("&lt;hyperlink&gt;")
assert "<hyperlink>" not in title
assert "](javascript:" not in markdown
assert "javascript:alert(1)" not in relationships
def test_odp_skips_hidden_drawing_page_styles_in_output_and_metadata() -> None:
"""验证 ODP converter 与 metadata 共用 drawing-page 可见性解析。"""
content = """<office:document-content
xmlns:office="urn:oasis:names:tc:opendocument:xmlns:office:1.0"
xmlns:style="urn:oasis:names:tc:opendocument:xmlns:style:1.0"
xmlns:presentation="urn:oasis:names:tc:opendocument:xmlns:presentation:1.0"
xmlns:draw="urn:oasis:names:tc:opendocument:xmlns:drawing:1.0"
xmlns:text="urn:oasis:names:tc:opendocument:xmlns:text:1.0">
<office:automatic-styles>
<style:style style:name="HiddenPage" style:family="drawing-page">
<style:drawing-page-properties presentation:visibility="hidden"/>
</style:style>
</office:automatic-styles>
<office:body><office:presentation>
<draw:page draw:name="Visible"><draw:frame><draw:text-box>
<text:p>visible slide</text:p>
</draw:text-box></draw:frame></draw:page>
<draw:page draw:name="StyledHidden" draw:style-name="HiddenPage"><draw:frame><draw:text-box>
<text:p>styled hidden slide</text:p>
</draw:text-box></draw:frame></draw:page>
<draw:page draw:name="DirectHidden" presentation:visibility="hidden"><draw:frame><draw:text-box>
<text:p>direct hidden slide</text:p>
</draw:text-box></draw:frame></draw:page>
</office:presentation></office:body></office:document-content>"""
payload = build_odf_package("odp", content)
pages = OdpModel().predict(BytesIO(payload))
metadata = extract_odf_metadata(BytesIO(payload), "odp")
assert len(pages) == 1
assert "visible slide" in str(pages)
assert "hidden slide" not in str(pages)
assert metadata["page_count"] == 1
def test_odf_style_cycle_is_bounded_and_preserves_text() -> None:
"""验证循环 parent-style-name 在有限链路内降级,不阻塞正文解析。"""
content = """<office:document-content