feat: implement API request handling with file parsing options and validation

This commit is contained in:
myhloli
2026-05-16 14:00:36 +08:00
parent 6d05b8895a
commit 7b1a98dc91
3 changed files with 234 additions and 194 deletions
+197
View File
@@ -0,0 +1,197 @@
# Copyright (c) Opendatalab. All rights reserved.
from dataclasses import dataclass
from typing import Annotated, Optional
from fastapi import File, Form, HTTPException, Request, UploadFile
from mineru.cli.public_http_client_policy import validate_public_http_client_request
ALLOWED_PARSE_METHODS = {"auto", "txt", "ocr"}
SWAGGER_UI_FILE_ARRAY_SCHEMA_EXTRA = {
# Swagger UI 5 currently fails to render a usable multi-file picker when
# FastAPI emits OpenAPI 3.1 byte arrays with contentMediaType.
"items": {"type": "string", "format": "binary"}
}
@dataclass
class ParseRequestOptions:
"""保存公开解析接口共用的表单参数,供 API 与 Router 复用。"""
files: list[UploadFile]
lang_list: list[str]
backend: str
parse_method: str
formula_enable: bool
table_enable: bool
image_analysis: bool
server_url: Optional[str]
return_md: bool
return_middle_json: bool
return_model_output: bool
return_content_list: bool
return_images: bool
response_format_zip: bool
return_original_file: bool
start_page_id: int
end_page_id: int
def validate_parse_method(parse_method: str) -> str:
"""校验公开 API 允许的 PDF 解析方式,避免各入口维护不同规则。"""
if parse_method not in ALLOWED_PARSE_METHODS:
raise HTTPException(
status_code=400,
detail=(
"Invalid parse_method. Allowed values: "
+ ", ".join(sorted(ALLOWED_PARSE_METHODS))
),
)
return parse_method
async def parse_request_form(
request: Request,
files: Annotated[
list[UploadFile],
File(
description="Upload PDF, image, DOCX, PPTX, or XLSX files for parsing",
json_schema_extra=SWAGGER_UI_FILE_ARRAY_SCHEMA_EXTRA,
),
],
lang_list: Annotated[
list[str],
Form(
description="""(Adapted only for pipeline and hybrid backend)Input the languages in the pdf to improve OCR accuracy.Options:
- ch: Chinese, English, Chinese Traditional.
- ch_lite: Chinese, English, Chinese Traditional, Japanese.
- ch_server: Chinese, English, Chinese Traditional, Japanese.
- en: English.
- korean: Korean, English.
- japan: Chinese, English, Chinese Traditional, Japanese.
- chinese_cht: Chinese, English, Chinese Traditional, Japanese.
- ta: Tamil, English.
- te: Telugu, English.
- ka: Kannada.
- th: Thai, English.
- el: Greek, English.
- latin: French, German, Afrikaans, Italian, Spanish, Bosnian, Portuguese, Czech, Welsh, Danish, Estonian, Irish, Croatian, Uzbek, Hungarian, Serbian (Latin), Indonesian, Occitan, Icelandic, Lithuanian, Maori, Malay, Dutch, Norwegian, Polish, Slovak, Slovenian, Albanian, Swedish, Swahili, Tagalog, Turkish, Latin, Azerbaijani, Kurdish, Latvian, Maltese, Pali, Romanian, Vietnamese, Finnish, Basque, Galician, Luxembourgish, Romansh, Catalan, Quechua.
- arabic: Arabic, Persian, Uyghur, Urdu, Pashto, Kurdish, Sindhi, Balochi, English.
- east_slavic: Russian, Belarusian, Ukrainian, English.
- cyrillic: Russian, Belarusian, Ukrainian, Serbian (Cyrillic), Bulgarian, Mongolian, Abkhazian, Adyghe, Kabardian, Avar, Dargin, Ingush, Chechen, Lak, Lezgin, Tabasaran, Kazakh, Kyrgyz, Tajik, Macedonian, Tatar, Chuvash, Bashkir, Malian, Moldovan, Udmurt, Komi, Ossetian, Buryat, Kalmyk, Tuvan, Sakha, Karakalpak, English.
- devanagari: Hindi, Marathi, Nepali, Bihari, Maithili, Angika, Bhojpuri, Magahi, Santali, Newari, Konkani, Sanskrit, Haryanvi, English.
""",
),
] = ["ch"],
backend: Annotated[
str,
Form(
description="""The backend for parsing:
- pipeline: More general, supports multiple languages, hallucination-free.
- vlm-auto-engine: High accuracy via local computing power, supports Chinese and English documents only.
- vlm-http-client: High accuracy via remote computing power(client suitable for openai-compatible servers), supports Chinese and English documents only.
- hybrid-auto-engine: Next-generation high accuracy solution via local computing power, supports multiple languages.
- hybrid-http-client: High accuracy via remote computing power but requires a little local computing power(client suitable for openai-compatible servers), supports multiple languages.""",
),
] = "hybrid-auto-engine",
parse_method: Annotated[
str,
Form(
description="""(Adapted only for pipeline and hybrid backend)The method for parsing PDF:
- auto: Automatically determine the method based on the file type
- txt: Use text extraction method
- ocr: Use OCR method for image-based PDFs
""",
),
] = "auto",
formula_enable: Annotated[
bool,
Form(description="Enable formula parsing."),
] = True,
table_enable: Annotated[
bool,
Form(description="Enable table parsing."),
] = True,
image_analysis: Annotated[
bool,
Form(description="Enable image/chart analysis for VLM and hybrid backends."),
] = True,
server_url: Annotated[
Optional[str],
Form(
description="(Adapted only for <vlm/hybrid>-http-client backend)openai compatible server url, e.g., http://127.0.0.1:30000",
),
] = None,
return_md: Annotated[
bool,
Form(description="Return markdown content in response"),
] = True,
return_middle_json: Annotated[
bool,
Form(description="Return middle JSON in response"),
] = False,
return_model_output: Annotated[
bool,
Form(description="Return model output JSON in response"),
] = False,
return_content_list: Annotated[
bool,
Form(description="Return content list JSON in response"),
] = False,
return_images: Annotated[
bool,
Form(description="Return extracted images in response"),
] = False,
response_format_zip: Annotated[
bool,
Form(description="Return results as a ZIP file instead of JSON"),
] = False,
return_original_file: Annotated[
bool,
Form(
description=(
"Include the processed original input file in the ZIP result; "
"ignored unless response_format_zip=true"
),
),
] = False,
start_page_id: Annotated[
int,
Form(description="The starting page for PDF parsing, beginning from 0"),
] = 0,
end_page_id: Annotated[
int,
Form(description="The ending page for PDF parsing, beginning from 0"),
] = 99999,
) -> ParseRequestOptions:
"""解析 API/Router 共用的 multipart 表单,并保持 Swagger 参数同源。"""
validate_public_http_client_request(
public_bind_exposed=bool(
getattr(request.app.state, "public_bind_exposed", False)
),
allow_public_http_client=bool(
getattr(request.app.state, "allow_public_http_client", False)
),
backend=backend,
server_url=server_url,
)
effective_return_original_file = return_original_file and response_format_zip
return ParseRequestOptions(
files=files,
lang_list=lang_list,
backend=backend,
parse_method=validate_parse_method(parse_method),
formula_enable=formula_enable,
table_enable=table_enable,
image_analysis=image_analysis,
server_url=server_url,
return_md=return_md,
return_middle_json=return_middle_json,
return_model_output=return_model_output,
return_content_list=return_content_list,
return_images=return_images,
response_format_zip=response_format_zip,
return_original_file=effective_return_original_file,
start_page_id=start_page_id,
end_page_id=end_page_id,
)
+1 -188
View File
@@ -21,8 +21,6 @@ from fastapi import (
BackgroundTasks,
Depends,
FastAPI,
File,
Form,
HTTPException,
Request,
UploadFile,
@@ -44,10 +42,10 @@ from mineru.cli.common import (
read_fn,
uniquify_task_stems,
)
from mineru.cli.api_request import ParseRequestOptions, parse_request_form
from mineru.cli.public_http_client_policy import (
configure_public_http_client_policy,
is_public_bind_host,
validate_public_http_client_request,
warn_if_public_http_client_policy as _warn_if_public_http_client_policy,
)
from mineru.cli.output_paths import resolve_parse_dir
@@ -86,18 +84,12 @@ RESULT_IMAGE_SUFFIXES = set(image_suffixes) | {"svg"}
DEFAULT_TASK_RETENTION_SECONDS = 24 * 60 * 60
DEFAULT_TASK_CLEANUP_INTERVAL_SECONDS = 5 * 60
DEFAULT_OUTPUT_ROOT = "./output"
ALLOWED_PARSE_METHODS = {"auto", "txt", "ocr"}
FILE_PARSE_TASK_ID_HEADER = "X-MinerU-Task-Id"
FILE_PARSE_TASK_STATUS_HEADER = "X-MinerU-Task-Status"
FILE_PARSE_TASK_STATUS_URL_HEADER = "X-MinerU-Task-Status-Url"
FILE_PARSE_TASK_RESULT_URL_HEADER = "X-MinerU-Task-Result-Url"
MINERU_API_PUBLIC_BIND_EXPOSED_ENV = "MINERU_API_PUBLIC_BIND_EXPOSED"
MINERU_API_ALLOW_PUBLIC_HTTP_CLIENT_ENV = "MINERU_API_ALLOW_PUBLIC_HTTP_CLIENT"
SWAGGER_UI_FILE_ARRAY_SCHEMA_EXTRA = {
# Swagger UI 5 currently fails to render a usable multi-file picker when
# FastAPI emits OpenAPI 3.1 byte arrays with contentMediaType.
"items": {"type": "string", "format": "binary"}
}
# 并发控制器
_request_semaphore: Optional[asyncio.Semaphore] = None
@@ -138,27 +130,6 @@ def install_stdin_shutdown_watcher(server: uvicorn.Server) -> None:
watcher.start()
@dataclass
class ParseRequestOptions:
files: list[UploadFile]
lang_list: list[str]
backend: str
parse_method: str
formula_enable: bool
table_enable: bool
image_analysis: bool
server_url: Optional[str]
return_md: bool
return_middle_json: bool
return_model_output: bool
return_content_list: bool
return_images: bool
response_format_zip: bool
return_original_file: bool
start_page_id: int
end_page_id: int
@dataclass
class StoredUpload:
original_name: str
@@ -374,18 +345,6 @@ def warn_if_public_http_client_policy(host: str, allow_public_http_client: bool)
)
def validate_parse_method(parse_method: str) -> str:
if parse_method not in ALLOWED_PARSE_METHODS:
raise HTTPException(
status_code=400,
detail=(
"Invalid parse_method. Allowed values: "
+ ", ".join(sorted(ALLOWED_PARSE_METHODS))
),
)
return parse_method
def cleanup_file(file_path: str) -> None:
"""清理临时文件或目录"""
try:
@@ -782,152 +741,6 @@ async def build_sync_file_parse_response(
)
async def parse_request_form(
request: Request,
files: Annotated[
list[UploadFile],
File(
description="Upload PDF, image, DOCX, PPTX, or XLSX files for parsing",
json_schema_extra=SWAGGER_UI_FILE_ARRAY_SCHEMA_EXTRA,
),
],
lang_list: Annotated[
list[str],
Form(
description="""(Adapted only for pipeline and hybrid backend)Input the languages in the pdf to improve OCR accuracy.Options:
- ch: Chinese, English, Chinese Traditional.
- ch_lite: Chinese, English, Chinese Traditional, Japanese.
- ch_server: Chinese, English, Chinese Traditional, Japanese.
- en: English.
- korean: Korean, English.
- japan: Chinese, English, Chinese Traditional, Japanese.
- chinese_cht: Chinese, English, Chinese Traditional, Japanese.
- ta: Tamil, English.
- te: Telugu, English.
- ka: Kannada.
- th: Thai, English.
- el: Greek, English.
- latin: French, German, Afrikaans, Italian, Spanish, Bosnian, Portuguese, Czech, Welsh, Danish, Estonian, Irish, Croatian, Uzbek, Hungarian, Serbian (Latin), Indonesian, Occitan, Icelandic, Lithuanian, Maori, Malay, Dutch, Norwegian, Polish, Slovak, Slovenian, Albanian, Swedish, Swahili, Tagalog, Turkish, Latin, Azerbaijani, Kurdish, Latvian, Maltese, Pali, Romanian, Vietnamese, Finnish, Basque, Galician, Luxembourgish, Romansh, Catalan, Quechua.
- arabic: Arabic, Persian, Uyghur, Urdu, Pashto, Kurdish, Sindhi, Balochi, English.
- east_slavic: Russian, Belarusian, Ukrainian, English.
- cyrillic: Russian, Belarusian, Ukrainian, Serbian (Cyrillic), Bulgarian, Mongolian, Abkhazian, Adyghe, Kabardian, Avar, Dargin, Ingush, Chechen, Lak, Lezgin, Tabasaran, Kazakh, Kyrgyz, Tajik, Macedonian, Tatar, Chuvash, Bashkir, Malian, Moldovan, Udmurt, Komi, Ossetian, Buryat, Kalmyk, Tuvan, Sakha, Karakalpak, English.
- devanagari: Hindi, Marathi, Nepali, Bihari, Maithili, Angika, Bhojpuri, Magahi, Santali, Newari, Konkani, Sanskrit, Haryanvi, English.
""",
),
] = ["ch"],
backend: Annotated[
str,
Form(
description="""The backend for parsing:
- pipeline: More general, supports multiple languages, hallucination-free.
- vlm-auto-engine: High accuracy via local computing power, supports Chinese and English documents only.
- vlm-http-client: High accuracy via remote computing power(client suitable for openai-compatible servers), supports Chinese and English documents only.
- hybrid-auto-engine: Next-generation high accuracy solution via local computing power, supports multiple languages.
- hybrid-http-client: High accuracy via remote computing power but requires a little local computing power(client suitable for openai-compatible servers), supports multiple languages.""",
),
] = "hybrid-auto-engine",
parse_method: Annotated[
str,
Form(
description="""(Adapted only for pipeline and hybrid backend)The method for parsing PDF:
- auto: Automatically determine the method based on the file type
- txt: Use text extraction method
- ocr: Use OCR method for image-based PDFs
""",
),
] = "auto",
formula_enable: Annotated[
bool,
Form(description="Enable formula parsing."),
] = True,
table_enable: Annotated[
bool,
Form(description="Enable table parsing."),
] = True,
image_analysis: Annotated[
bool,
Form(description="Enable image/chart analysis for VLM and hybrid backends."),
] = True,
server_url: Annotated[
Optional[str],
Form(
description="(Adapted only for <vlm/hybrid>-http-client backend)openai compatible server url, e.g., http://127.0.0.1:30000",
),
] = None,
return_md: Annotated[
bool,
Form(description="Return markdown content in response"),
] = True,
return_middle_json: Annotated[
bool,
Form(description="Return middle JSON in response"),
] = False,
return_model_output: Annotated[
bool,
Form(description="Return model output JSON in response"),
] = False,
return_content_list: Annotated[
bool,
Form(description="Return content list JSON in response"),
] = False,
return_images: Annotated[
bool,
Form(description="Return extracted images in response"),
] = False,
response_format_zip: Annotated[
bool,
Form(description="Return results as a ZIP file instead of JSON"),
] = False,
return_original_file: Annotated[
bool,
Form(
description=(
"Include the processed original input file in the ZIP result; "
"ignored unless response_format_zip=true"
),
),
] = False,
start_page_id: Annotated[
int,
Form(description="The starting page for PDF parsing, beginning from 0"),
] = 0,
end_page_id: Annotated[
int,
Form(description="The ending page for PDF parsing, beginning from 0"),
] = 99999,
) -> ParseRequestOptions:
validate_public_http_client_request(
public_bind_exposed=bool(
getattr(request.app.state, "public_bind_exposed", False)
),
allow_public_http_client=bool(
getattr(request.app.state, "allow_public_http_client", False)
),
backend=backend,
server_url=server_url,
)
effective_return_original_file = return_original_file and response_format_zip
return ParseRequestOptions(
files=files,
lang_list=lang_list,
backend=backend,
parse_method=validate_parse_method(parse_method),
formula_enable=formula_enable,
table_enable=table_enable,
image_analysis=image_analysis,
server_url=server_url,
return_md=return_md,
return_middle_json=return_middle_json,
return_model_output=return_model_output,
return_content_list=return_content_list,
return_images=return_images,
response_format_zip=response_format_zip,
return_original_file=effective_return_original_file,
start_page_id=start_page_id,
end_page_id=end_page_id,
)
async def save_upload_files(upload_dir: str, files: list[UploadFile]) -> list[StoredUpload]:
os.makedirs(upload_dir, exist_ok=True)
uploads: list[StoredUpload] = []
+36 -6
View File
@@ -13,12 +13,12 @@ from contextlib import ExitStack, asynccontextmanager, suppress
from dataclasses import dataclass, field
from datetime import datetime, timezone
from pathlib import Path
from typing import Any, Optional, Sequence
from typing import Annotated, Any, Optional, Sequence
import click
import httpx
import uvicorn
from fastapi import FastAPI, HTTPException, Request
from fastapi import Depends, FastAPI, HTTPException, Request
from fastapi.middleware.gzip import GZipMiddleware
from fastapi.responses import JSONResponse, Response, StreamingResponse
from loguru import logger
@@ -41,6 +41,7 @@ from mineru.cli.api_client import (
response_detail,
)
from mineru.cli.api_protocol import API_PROTOCOL_VERSION
from mineru.cli.api_request import ParseRequestOptions, parse_request_form
from mineru.cli.common import normalize_upload_filename
from mineru.cli.public_http_client_policy import (
configure_public_http_client_policy,
@@ -1455,8 +1456,22 @@ def create_app(settings: RouterSettings | None = None) -> FastAPI:
)
app.add_middleware(GZipMiddleware, minimum_size=1000)
@app.post(path="/tasks", status_code=202)
async def submit_parse_task(http_request: Request):
@app.post(
path="/tasks",
status_code=202,
summary="Submit an asynchronous parse task through the router",
description=(
"Submit files and parse options to a healthy upstream MinerU API "
"server selected by the router, then return a router task id."
),
)
async def submit_parse_task(
http_request: Request,
request_options: Annotated[
ParseRequestOptions, Depends(parse_request_form)
],
):
del request_options
payload = await stage_multipart_request(http_request)
try:
router_task = await submit_router_task(http_request, payload)
@@ -1501,8 +1516,23 @@ def create_app(settings: RouterSettings | None = None) -> FastAPI:
)
return await proxy_router_task_result(request, task)
@app.post(path="/file_parse", status_code=200)
async def file_parse(request: Request):
@app.post(
path="/file_parse",
status_code=200,
summary="Synchronously parse uploaded files through the router",
description=(
"Submit files and parse options to a healthy upstream MinerU API "
"server selected by the router, wait for completion, and proxy the "
"final result in the same response."
),
)
async def file_parse(
request: Request,
request_options: Annotated[
ParseRequestOptions, Depends(parse_request_form)
],
):
del request_options
payload = await stage_multipart_request(request)
try:
router_task = await submit_router_task(request, payload)