mirror of
https://github.com/opendatalab/MinerU.git
synced 2026-08-30 17:12:39 +08:00
fix: harden native parser output boundaries
This commit is contained in:
@@ -20,6 +20,7 @@ MAX_CSV_ROWS: Final = 1_048_576
|
||||
MAX_CSV_COLUMNS: Final = 16_384
|
||||
# CSV 会把每个槽位实体化为 HTML/DOM 节点,预算需显著低于稀疏电子表格投影上限。
|
||||
MAX_CSV_GRID_SLOTS: Final = 250_000
|
||||
MAX_CSV_RENDERED_BYTES: Final = 256 * 1024 * 1024
|
||||
|
||||
_DELIMITER_CANDIDATES: Final = (",", ";", "\t", "|")
|
||||
_DELIMITER_SAMPLE_RECORDS: Final = 20
|
||||
@@ -255,17 +256,78 @@ def _render_field_html(value: str) -> str:
|
||||
return html.escape(normalized, quote=True).replace("\n", "<br>")
|
||||
|
||||
|
||||
def _rendered_field_utf8_bytes(value: str, remaining_budget: int) -> int:
|
||||
"""在不创建转义字符串的前提下计算字段渲染后的 UTF-8 字节数。"""
|
||||
rendered_bytes = 0
|
||||
index = 0
|
||||
while index < len(value):
|
||||
char = value[index]
|
||||
codepoint = ord(char)
|
||||
if char == "\r":
|
||||
if index + 1 < len(value) and value[index + 1] == "\n":
|
||||
index += 1
|
||||
addition = len("<br>")
|
||||
elif char == "\n":
|
||||
addition = len("<br>")
|
||||
elif codepoint <= 0x08 or codepoint in {0x0B, 0x0C, 0x7F} or 0x0E <= codepoint <= 0x1F:
|
||||
addition = 3
|
||||
elif 0xD800 <= codepoint <= 0xDFFF:
|
||||
addition = 3
|
||||
elif char == "&":
|
||||
addition = len("&")
|
||||
elif char in {"<", ">"}:
|
||||
addition = len("<")
|
||||
elif char in {'"', "'"}:
|
||||
addition = len(""")
|
||||
elif codepoint <= 0x7F:
|
||||
addition = 1
|
||||
elif codepoint <= 0x7FF:
|
||||
addition = 2
|
||||
elif codepoint <= 0xFFFF:
|
||||
addition = 3
|
||||
else:
|
||||
addition = 4
|
||||
rendered_bytes += addition
|
||||
if rendered_bytes > remaining_budget:
|
||||
raise ValueError(f"CSV exceeds max_rendered_bytes={MAX_CSV_RENDERED_BYTES}")
|
||||
index += 1
|
||||
return rendered_bytes
|
||||
|
||||
|
||||
def _charge_rendered_bytes(used_bytes: int, additional_bytes: int) -> int:
|
||||
"""累计 CSV HTML 输出预算,并在写入 StringIO 前拒绝超限内容。"""
|
||||
if additional_bytes < 0 or used_bytes > MAX_CSV_RENDERED_BYTES - additional_bytes:
|
||||
raise ValueError(f"CSV exceeds max_rendered_bytes={MAX_CSV_RENDERED_BYTES}")
|
||||
return used_bytes + additional_bytes
|
||||
|
||||
|
||||
def _rows_to_html(rows: list[list[str]], *, has_header: bool) -> str:
|
||||
"""增量构造安全表格 HTML,避免为每个单元格保留独立字符串对象。"""
|
||||
output = StringIO()
|
||||
rendered_bytes = 0
|
||||
rendered_bytes = _charge_rendered_bytes(rendered_bytes, len("<table>"))
|
||||
output.write("<table>")
|
||||
for row_index, row in enumerate(rows):
|
||||
tag = "th" if has_header and row_index == 0 else "td"
|
||||
output.write("\n <tr>")
|
||||
row_prefix = "\n <tr>"
|
||||
rendered_bytes = _charge_rendered_bytes(rendered_bytes, len(row_prefix))
|
||||
output.write(row_prefix)
|
||||
for value in row:
|
||||
output.write(f"\n <{tag}>{_render_field_html(value)}</{tag}>")
|
||||
output.write("\n </tr>")
|
||||
output.write("\n</table>")
|
||||
cell_prefix = f"\n <{tag}>"
|
||||
cell_suffix = f"</{tag}>"
|
||||
rendered_bytes = _charge_rendered_bytes(rendered_bytes, len(cell_prefix) + len(cell_suffix))
|
||||
remaining_budget = MAX_CSV_RENDERED_BYTES - rendered_bytes
|
||||
field_bytes = _rendered_field_utf8_bytes(value, remaining_budget)
|
||||
rendered_bytes = _charge_rendered_bytes(rendered_bytes, field_bytes)
|
||||
output.write(cell_prefix)
|
||||
output.write(_render_field_html(value))
|
||||
output.write(cell_suffix)
|
||||
row_suffix = "\n </tr>"
|
||||
rendered_bytes = _charge_rendered_bytes(rendered_bytes, len(row_suffix))
|
||||
output.write(row_suffix)
|
||||
table_suffix = "\n</table>"
|
||||
_charge_rendered_bytes(rendered_bytes, len(table_suffix))
|
||||
output.write(table_suffix)
|
||||
return output.getvalue()
|
||||
|
||||
|
||||
|
||||
@@ -489,7 +489,7 @@ class EpubChapterConverter:
|
||||
if name == "figure":
|
||||
return self._parse_figure(element, resolved.text)
|
||||
if name == "svg":
|
||||
return self._parse_svg(element)
|
||||
return self._parse_svg(element, resolved.text)
|
||||
return self._parse_container_contents(element, resolved.text)
|
||||
|
||||
def _parse_note_element(self, element: etree._Element, style: TextStyle) -> list[dict[str, object]]:
|
||||
@@ -622,20 +622,44 @@ class EpubChapterConverter:
|
||||
blocks.append({"type": BlockType.TEXT, "content": html.escape(caption, quote=False)})
|
||||
return blocks
|
||||
|
||||
def _parse_svg(self, element: etree._Element) -> list[dict[str, object]]:
|
||||
def _visible_svg_text(self, element: etree._Element, style: TextStyle) -> str:
|
||||
"""递归提取 SVG 可见文本,并排除隐藏后代的内容。"""
|
||||
parts = [_clean_text_node(element.text)]
|
||||
for child in element:
|
||||
if not isinstance(child.tag, str):
|
||||
parts.append(_clean_text_node(_entity_text(child)))
|
||||
parts.append(_clean_text_node(child.tail))
|
||||
continue
|
||||
resolved = self.stylesheet.resolve(child, style)
|
||||
if not resolved.hidden:
|
||||
parts.append(self._visible_svg_text(child, resolved.text))
|
||||
parts.append(_clean_text_node(child.tail))
|
||||
return _WHITESPACE_RE.sub(" ", html.unescape("".join(parts))).strip()
|
||||
|
||||
def _parse_svg(self, element: etree._Element, style: TextStyle) -> list[dict[str, object]]:
|
||||
"""从 SVG 尽力提取 title/desc/text 和包内栅格 image。"""
|
||||
blocks: list[dict[str, object]] = []
|
||||
texts: list[str] = []
|
||||
for child in element.iter():
|
||||
if not isinstance(child.tag, str):
|
||||
continue
|
||||
name = _local_name(child)
|
||||
if name in {"title", "desc", "text"}:
|
||||
value = _visible_text(child)
|
||||
if value and value not in texts:
|
||||
texts.append(value)
|
||||
elif name == "image":
|
||||
blocks.extend(self._image_blocks(child))
|
||||
|
||||
def visit(parent: etree._Element, inherited: TextStyle) -> None:
|
||||
"""按 SVG 树顺序访问可见候选节点,并让祖先隐藏状态截断子树。"""
|
||||
for child in parent:
|
||||
if not isinstance(child.tag, str):
|
||||
continue
|
||||
resolved = self.stylesheet.resolve(child, inherited)
|
||||
if resolved.hidden:
|
||||
continue
|
||||
name = _local_name(child)
|
||||
if name in {"title", "desc", "text"}:
|
||||
value = self._visible_svg_text(child, resolved.text)
|
||||
if value and value not in texts:
|
||||
texts.append(value)
|
||||
elif name == "image":
|
||||
blocks.extend(self._image_blocks(child))
|
||||
else:
|
||||
visit(child, resolved.text)
|
||||
|
||||
visit(element, style)
|
||||
if texts:
|
||||
blocks.insert(0, {"type": BlockType.TEXT, "content": html.escape("\n".join(texts), quote=False)})
|
||||
return blocks
|
||||
@@ -880,7 +904,8 @@ def convert_svg_spine(
|
||||
"""把 standalone SVG spine item 尽力转换为文本和包内栅格图片。"""
|
||||
empty_registry = EpubAnchorRegistry([], package)
|
||||
converter = EpubChapterConverter(package, chapter_path, root, empty_registry)
|
||||
return converter._parse_svg(root)
|
||||
resolved = converter.stylesheet.resolve(root, TextStyle())
|
||||
return [] if resolved.hidden else converter._parse_svg(root, resolved.text)
|
||||
|
||||
|
||||
__all__ = [
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, BinaryIO, Iterator
|
||||
@@ -251,6 +252,8 @@ def _parse_odp_pages(context: _OdfContext) -> list[list[dict[str, Any]]]:
|
||||
for page in context.body:
|
||||
if page.tag != qname("draw", "page"):
|
||||
continue
|
||||
if not context.styles.drawing_page_is_visible(page):
|
||||
continue
|
||||
positioned: list[_PositionedBlocks] = []
|
||||
for order, (shape, x, y) in enumerate(_iter_slide_shapes(page)):
|
||||
presentation_class = shape.get(qname("presentation", "class"), "")
|
||||
@@ -324,7 +327,7 @@ def _parse_ods_pages(context: _OdfContext) -> list[list[dict[str, Any]]]:
|
||||
continue
|
||||
if not context.styles.table_is_visible(sheet.get(qname("table", "style-name"))):
|
||||
continue
|
||||
name = sheet.get(qname("table", "name"), "Sheet")
|
||||
name = html.escape(sheet.get(qname("table", "name"), "Sheet"), quote=False)
|
||||
sheet_pages.append((name, _sheet_blocks(sheet, parser)))
|
||||
if sum(bool(blocks) for _, blocks in sheet_pages) > 1:
|
||||
for name, blocks in sheet_pages:
|
||||
|
||||
@@ -73,7 +73,9 @@ def extract_odf_metadata(file_binary: BinaryIO, suffix: OdfSuffix) -> dict[str,
|
||||
if suffix == "odt":
|
||||
page_count = _odt_page_count(meta_root)
|
||||
elif suffix == "odp":
|
||||
page_count = sum(1 for child in body if child.tag == qname("draw", "page"))
|
||||
page_count = sum(
|
||||
1 for child in body if child.tag == qname("draw", "page") and styles.drawing_page_is_visible(child)
|
||||
)
|
||||
else:
|
||||
page_count = _visible_sheet_count(body, styles)
|
||||
return {
|
||||
|
||||
@@ -23,6 +23,7 @@ class _StyleDefinition:
|
||||
text_delta: TextStyleDelta
|
||||
master_page_name: str | None
|
||||
table_display: bool | None
|
||||
drawing_page_visible: bool | None
|
||||
|
||||
|
||||
class OdfStyles:
|
||||
@@ -35,6 +36,7 @@ class OdfStyles:
|
||||
self._list_styles: dict[str, dict[int, ListLevel]] = {}
|
||||
self._resolved_text: dict[tuple[str, str], TextStyleDelta] = {}
|
||||
self._resolved_table_display: dict[str, bool | None] = {}
|
||||
self._resolved_drawing_page_visibility: dict[str, bool | None] = {}
|
||||
self._master_pages: dict[str, etree._Element] = {}
|
||||
for root in roots:
|
||||
if root is not None:
|
||||
@@ -59,6 +61,7 @@ class OdfStyles:
|
||||
text_delta=self._text_delta(style),
|
||||
master_page_name=style.get(qname("style", "master-page-name")),
|
||||
table_display=self._table_display(style),
|
||||
drawing_page_visible=self._drawing_page_visibility(style),
|
||||
)
|
||||
for list_style in root.iter(qname("text", "list-style")):
|
||||
name = list_style.get(qname("style", "name"))
|
||||
@@ -133,6 +136,17 @@ class OdfStyles:
|
||||
return None
|
||||
return display.casefold() != "false"
|
||||
|
||||
@staticmethod
|
||||
def _drawing_page_visibility(style: etree._Element) -> bool | None:
|
||||
"""读取 drawing-page 样式的 presentation visibility。"""
|
||||
properties = style.find(qname("style", "drawing-page-properties"))
|
||||
if properties is None:
|
||||
return None
|
||||
visibility = properties.get(qname("presentation", "visibility"))
|
||||
if visibility is None:
|
||||
return None
|
||||
return visibility.casefold() != "hidden"
|
||||
|
||||
@staticmethod
|
||||
def _parse_list_style(element: etree._Element) -> dict[int, ListLevel]:
|
||||
"""解析列表样式的层级、类型和通用起始值。"""
|
||||
@@ -258,6 +272,34 @@ class OdfStyles:
|
||||
self._resolved_table_display[style_name] = resolved
|
||||
return resolved is not False
|
||||
|
||||
def drawing_page_is_visible(self, page: etree._Element) -> bool:
|
||||
"""解析 ODP 页面直接属性或 drawing-page 样式中的隐藏状态。"""
|
||||
direct_visibility = page.get(qname("presentation", "visibility"))
|
||||
if direct_visibility is not None:
|
||||
return direct_visibility.casefold() != "hidden"
|
||||
style_name = page.get(qname("draw", "style-name"))
|
||||
if not style_name:
|
||||
return True
|
||||
if style_name in self._resolved_drawing_page_visibility:
|
||||
return self._resolved_drawing_page_visibility[style_name] is not False
|
||||
seen: set[str] = set()
|
||||
current = style_name
|
||||
resolved: bool | None = None
|
||||
while current:
|
||||
if current in seen:
|
||||
logger.warning("ODF style inheritance cycle detected: family=drawing-page, style={}", current)
|
||||
break
|
||||
seen.add(current)
|
||||
definition = self._styles.get(("drawing-page", current))
|
||||
if definition is None:
|
||||
break
|
||||
if definition.drawing_page_visible is not None:
|
||||
resolved = definition.drawing_page_visible
|
||||
break
|
||||
current = definition.parent or ""
|
||||
self._resolved_drawing_page_visibility[style_name] = resolved
|
||||
return resolved is not False
|
||||
|
||||
def master_page(self, name: str | None) -> etree._Element | None:
|
||||
"""返回指定 master-page;空名称时优先使用第一个定义。"""
|
||||
if name and name in self._master_pages:
|
||||
|
||||
@@ -344,6 +344,15 @@ def _column_index(label: str) -> int | None:
|
||||
return result - 1 if result <= MAX_GRID_SLOTS else None
|
||||
|
||||
|
||||
def _row_index(label: str) -> int | None:
|
||||
"""在整数转换前把 A1 地址中的行号约束到共享网格预算。"""
|
||||
normalized = label.lstrip("0")
|
||||
if not normalized or len(normalized) > len(str(MAX_GRID_SLOTS)):
|
||||
return None
|
||||
result = int(normalized)
|
||||
return result - 1 if result <= MAX_GRID_SLOTS else None
|
||||
|
||||
|
||||
def parse_cell_range_bounds(address: str) -> tuple[int, int, int, int] | None:
|
||||
"""从 ODF cell-range-address 中提取零基闭区间边界。"""
|
||||
matches = list(_CELL_ADDRESS_RE.finditer(address or ""))
|
||||
@@ -351,11 +360,11 @@ def parse_cell_range_bounds(address: str) -> tuple[int, int, int, int] | None:
|
||||
return None
|
||||
first = matches[0]
|
||||
last = matches[-1]
|
||||
row_start = int(first.group("row")) - 1
|
||||
row_start = _row_index(first.group("row"))
|
||||
col_start = _column_index(first.group("col"))
|
||||
row_end = int(last.group("row")) - 1
|
||||
row_end = _row_index(last.group("row"))
|
||||
col_end = _column_index(last.group("col"))
|
||||
if col_start is None or col_end is None:
|
||||
if row_start is None or col_start is None or row_end is None or col_end is None:
|
||||
return None
|
||||
return min(row_start, row_end), max(row_start, row_end), min(col_start, col_end), max(col_start, col_end)
|
||||
|
||||
|
||||
@@ -146,6 +146,28 @@ def test_csv_grid_limit_short_circuits_before_trailing_malformed_record(monkeypa
|
||||
CsvModel().predict(BytesIO(b'a,b\n1,2\n"unterminated'))
|
||||
|
||||
|
||||
def test_csv_rendered_budget_fails_before_materializing_escaped_field(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
"""验证 HTML 展开超限时不会先创建放大的字段字符串。"""
|
||||
monkeypatch.setattr(csv_module, "MAX_CSV_RENDERED_BYTES", 64)
|
||||
|
||||
def unexpected_escape(_value: str) -> str:
|
||||
"""输出预算应在进入 html.escape 前拒绝字段。"""
|
||||
pytest.fail("oversized CSV field reached HTML escaping")
|
||||
|
||||
monkeypatch.setattr(csv_module, "_render_field_html", unexpected_escape)
|
||||
|
||||
with pytest.raises(ValueError, match="max_rendered_bytes"):
|
||||
csv_module._rows_to_html([["&" * 20]], has_header=False)
|
||||
|
||||
|
||||
def test_csv_rendered_size_estimator_matches_html_escape_semantics() -> None:
|
||||
"""验证特殊字符、换行、控制符和非 ASCII 文本的 UTF-8 预算精确。"""
|
||||
value = "&<>\"'\r\n\x01中"
|
||||
rendered = csv_module._render_field_html(value)
|
||||
|
||||
assert csv_module._rendered_field_utf8_bytes(value, 1_000) == len(rendered.encode())
|
||||
|
||||
|
||||
def test_csv_default_grid_budget_rejects_wide_dom_before_rendering() -> None:
|
||||
"""验证默认预算在宽空表生成数十万 HTML 节点前拒绝输入。"""
|
||||
assert csv_module.MAX_CSV_GRID_SLOTS == 250_000
|
||||
|
||||
@@ -22,7 +22,7 @@ from mineru.errors import InvalidRequestError
|
||||
from mineru.model.flash import EpubModel
|
||||
from mineru.model.flash.epub import EpubEncryptedError, EpubPackage, EpubParseError, EpubResourceLimitError, detect_epub
|
||||
from mineru.model.flash.epub.styles import EpubStylesheet, TextStyle
|
||||
from mineru.model.flash.epub.xhtml import EpubChapterConverter, build_anchor_registry
|
||||
from mineru.model.flash.epub.xhtml import EpubChapterConverter, build_anchor_registry, convert_svg_spine
|
||||
from mineru.parser import MinerUParser, parse, parse_async
|
||||
from mineru.parser import api_server
|
||||
from mineru.parser.api_server import CreateJobRequest, FileStore
|
||||
@@ -502,6 +502,35 @@ def test_epub_corrupt_chapter_keeps_empty_spine_placeholder() -> None:
|
||||
assert "SVG text" in middle.pages[2].blocks[0].content # type: ignore[union-attr]
|
||||
|
||||
|
||||
def test_epub_svg_extraction_skips_hidden_descendants_and_hidden_root() -> None:
|
||||
"""验证 standalone SVG 不提取隐藏文本、隐藏图片或隐藏祖先子树。"""
|
||||
package = EpubPackage(build_epub_fixture())
|
||||
path = "EPUB/fixed/page.svg"
|
||||
try:
|
||||
root = package.xml_part(path)
|
||||
assert root is not None
|
||||
namespace = etree.QName(root).namespace
|
||||
original_text = next(root.iter(f"{{{namespace}}}text"))
|
||||
original_text.set("style", "display: none")
|
||||
original_image = next(root.iter(f"{{{namespace}}}image"))
|
||||
original_image.set("style", "visibility: hidden")
|
||||
hidden_group = etree.SubElement(root, f"{{{namespace}}}g", style="display: none")
|
||||
etree.SubElement(hidden_group, f"{{{namespace}}}text").text = "hidden group text"
|
||||
etree.SubElement(root, f"{{{namespace}}}text").text = "visible graphic label"
|
||||
|
||||
blocks = convert_svg_spine(package, path, root)
|
||||
|
||||
assert "visible graphic label" in str(blocks)
|
||||
assert "SVG text" not in str(blocks)
|
||||
assert "hidden group text" not in str(blocks)
|
||||
assert all(block["type"] != BlockType.IMAGE for block in blocks)
|
||||
|
||||
root.set("style", "display: none")
|
||||
assert convert_svg_spine(package, path, root) == []
|
||||
finally:
|
||||
package.close()
|
||||
|
||||
|
||||
def test_epub_malformed_resource_and_link_references_degrade_locally() -> None:
|
||||
"""验证非法 URI 只丢弃样式、图片或链接目标,不阻断章节正文。"""
|
||||
package = EpubPackage(build_epub_fixture())
|
||||
|
||||
@@ -582,6 +582,15 @@ def test_odf_rejects_overlong_chart_columns_before_bigint_conversion(monkeypatch
|
||||
assert odf_table_module.parse_cell_range_bounds(f"local-table.{'A' * 100_000}1") is None
|
||||
|
||||
|
||||
def test_odf_rejects_overlong_chart_rows_before_bigint_conversion(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
"""验证 chart A1 行号在 int 转换前受共享网格预算约束。"""
|
||||
monkeypatch.setattr(odf_table_module, "MAX_GRID_SLOTS", 4)
|
||||
|
||||
assert odf_table_module.parse_cell_range_bounds("local-table.A4:B4") == (3, 3, 0, 1)
|
||||
assert odf_table_module.parse_cell_range_bounds("local-table.A5:B5") is None
|
||||
assert odf_table_module.parse_cell_range_bounds(f"local-table.A{'9' * 100_000}") is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"target",
|
||||
[
|
||||
@@ -737,6 +746,70 @@ def test_odf_flattened_titles_and_notes_keep_literal_protocol_escaped() -> None:
|
||||
assert "javascript:alert(1)" not in relationships
|
||||
|
||||
|
||||
def test_ods_sheet_titles_escape_literal_inline_protocol() -> None:
|
||||
"""验证多 sheet 标题不会把名称中的内部协议重建为活动链接。"""
|
||||
literal = "<hyperlink><text>x</text><url>javascript:alert(1)</url></hyperlink>"
|
||||
content = f"""<office:document-content
|
||||
xmlns:office="urn:oasis:names:tc:opendocument:xmlns:office:1.0"
|
||||
xmlns:table="urn:oasis:names:tc:opendocument:xmlns:table:1.0"
|
||||
xmlns:text="urn:oasis:names:tc:opendocument:xmlns:text:1.0">
|
||||
<office:body><office:spreadsheet>
|
||||
<table:table table:name="{literal}"><table:table-row><table:table-cell>
|
||||
<text:p>A</text:p>
|
||||
</table:table-cell></table:table-row></table:table>
|
||||
<table:table table:name="Safe"><table:table-row><table:table-cell>
|
||||
<text:p>B</text:p>
|
||||
</table:table-cell></table:table-row></table:table>
|
||||
</office:spreadsheet></office:body></office:document-content>"""
|
||||
middle, model = doc_analyze(build_odf_package("ods", content), file_suffix="ods")
|
||||
|
||||
title = model.pages[0][0]["content"]
|
||||
markdown = render_markdown(middle)
|
||||
docx = render_docx(middle)
|
||||
with ZipFile(BytesIO(docx)) as package:
|
||||
relationships = package.read("word/_rels/document.xml.rels").decode("utf-8")
|
||||
|
||||
assert title.startswith("<hyperlink>")
|
||||
assert "<hyperlink>" not in title
|
||||
assert "](javascript:" not in markdown
|
||||
assert "javascript:alert(1)" not in relationships
|
||||
|
||||
|
||||
def test_odp_skips_hidden_drawing_page_styles_in_output_and_metadata() -> None:
|
||||
"""验证 ODP converter 与 metadata 共用 drawing-page 可见性解析。"""
|
||||
content = """<office:document-content
|
||||
xmlns:office="urn:oasis:names:tc:opendocument:xmlns:office:1.0"
|
||||
xmlns:style="urn:oasis:names:tc:opendocument:xmlns:style:1.0"
|
||||
xmlns:presentation="urn:oasis:names:tc:opendocument:xmlns:presentation:1.0"
|
||||
xmlns:draw="urn:oasis:names:tc:opendocument:xmlns:drawing:1.0"
|
||||
xmlns:text="urn:oasis:names:tc:opendocument:xmlns:text:1.0">
|
||||
<office:automatic-styles>
|
||||
<style:style style:name="HiddenPage" style:family="drawing-page">
|
||||
<style:drawing-page-properties presentation:visibility="hidden"/>
|
||||
</style:style>
|
||||
</office:automatic-styles>
|
||||
<office:body><office:presentation>
|
||||
<draw:page draw:name="Visible"><draw:frame><draw:text-box>
|
||||
<text:p>visible slide</text:p>
|
||||
</draw:text-box></draw:frame></draw:page>
|
||||
<draw:page draw:name="StyledHidden" draw:style-name="HiddenPage"><draw:frame><draw:text-box>
|
||||
<text:p>styled hidden slide</text:p>
|
||||
</draw:text-box></draw:frame></draw:page>
|
||||
<draw:page draw:name="DirectHidden" presentation:visibility="hidden"><draw:frame><draw:text-box>
|
||||
<text:p>direct hidden slide</text:p>
|
||||
</draw:text-box></draw:frame></draw:page>
|
||||
</office:presentation></office:body></office:document-content>"""
|
||||
payload = build_odf_package("odp", content)
|
||||
|
||||
pages = OdpModel().predict(BytesIO(payload))
|
||||
metadata = extract_odf_metadata(BytesIO(payload), "odp")
|
||||
|
||||
assert len(pages) == 1
|
||||
assert "visible slide" in str(pages)
|
||||
assert "hidden slide" not in str(pages)
|
||||
assert metadata["page_count"] == 1
|
||||
|
||||
|
||||
def test_odf_style_cycle_is_bounded_and_preserves_text() -> None:
|
||||
"""验证循环 parent-style-name 在有限链路内降级,不阻塞正文解析。"""
|
||||
content = """<office:document-content
|
||||
|
||||
Reference in New Issue
Block a user