feat: add MLU support and update documentation for Cambricon integration

This commit is contained in:
myhloli
2026-01-30 15:10:14 +08:00
parent 0d6211aa52
commit 3e5fa8770f
7 changed files with 161 additions and 244 deletions
+28
View File
@@ -0,0 +1,28 @@
# Base image containing the vLLM inference environment, requiring amd64(x86-64) CPU + Cambricon MLU.
FROM
# Install Noto fonts for Chinese characters
RUN apt-get update && \
apt-get install -y \
fonts-noto-core \
fonts-noto-cjk \
fontconfig && \
fc-cache -fv && \
apt-get clean && \
rm -rf /var/lib/apt/lists/*
# Install mineru latest
RUN python3 -m pip install -U pip -i https://mirrors.aliyun.com/pypi/simple && \
python3 -m pip install 'mineru[core]>=2.7.3' \
numpy==1.26.4 \
opencv-python==4.11.0.86 \
accelerate==1.2.0 \
-i https://mirrors.aliyun.com/pypi/simple && \
python3 -m pip cache purge
# Download models and update the configuration file
RUN /bin/bash -c "mineru-models-download -s modelscope -m all"
# Set the entry point to activate the virtual environment and run the command line tool
ENTRYPOINT ["/bin/bash", "-c", "export MINERU_MODEL_SOURCE=local && exec \"$@\"", "--"]
+116 -242
View File
@@ -1,253 +1,127 @@
# MinerU
## 1. 环境准备
容器启动方式见第3节
### 1.1 获取代码
## 1. 测试平台
以下为本指南测试使用的平台信息,供参考:
```
git clone https://github.com/opendatalab/MinerU.git
git checkout fa1149cd4abf9db5e0f13e4e074cdb568be189f4
```
### 1.2 安装依赖
```
source /torch/venv3/pytorch_infer/bin/activate
pip install accelerate==1.11.0 doclayout_yolo==0.0.4 thop==0.1.1.post2209072238 ultralytics-thop==2.0.18 ultralytics==8.3.228
# requirements_check.txt具体内容在下面
pip install -r requirements_check.txt
cd MinerU
pip install -e .[core] --no-deps
```
requirements_check.txt
```
# triton==3.0.0+mlu1.3.1
# torch==2.5.0+cpu
# torchvision==0.20.0+cpu
# === 1. 已安装且版本相同 ===
# (这些包已满足要求, 无需操作)
# === 2. 已安装但版本不同 ===
# (运行 pip install -r 将强制更新到左侧的目标版本)
# accelerate==1.11.0 # 0.33.0
beautifulsoup4==4.14.2 # 4.12.3
cffi==2.0.0 # 1.17.1
huggingface-hub==0.36.0 # 0.25.2
jiter==0.12.0 # 0.8.2
openai==2.8.0 # 1.59.7
pillow==11.3.0 # 10.4.0
sympy==1.14.0 # 1.13.1
tokenizers==0.22.1 # 0.21.0
# torch==2.9.1 # 2.5.0+cpu
# torchvision==0.24.1 # 0.20.0+cpu
transformers==4.57.1 # 4.48.0
# triton==3.5.1 # 3.0.0+mlu1.3.1
typing-extensions==4.15.0 # 4.12.2
# === 3. 未安装 ===
# (运行 pip install -r 将安装这些包)
aiofiles==24.1.0
albucore==0.0.24
albumentations==2.0.8
antlr4-python3-runtime==4.9.3
brotli==1.2.0
coloredlogs==15.0.1
colorlog==6.10.1
cryptography==46.0.3
# doclayout_yolo==0.0.4
fast-langdetect==0.2.5
fasttext-predict==0.9.2.4
ffmpy==1.0.0
flatbuffers==25.9.23
ftfy==6.3.1
gradio-client==1.13.3
gradio-pdf==0.0.22
gradio==5.49.1
groovy==0.1.2
hf-xet==1.2.0
httpx-retries==0.4.5
humanfriendly==10.0
imageio==2.37.2
json-repair==0.53.0
magika==0.6.3
markdown-it-py==4.0.0
mdurl==0.1.2
mineru-vl-utils==0.1.15
mineru==2.6.4
modelscope==1.31.0
# nvidia-cublas-cu12==12.8.4.1
# nvidia-cuda-cupti-cu12==12.8.90
# nvidia-cuda-nvrtc-cu12==12.8.93
# nvidia-cuda-runtime-cu12==12.8.90
# nvidia-cudnn-cu12==9.10.2.21
# nvidia-cufft-cu12==11.3.3.83
# nvidia-cufile-cu12==1.13.1.3
# nvidia-curand-cu12==10.3.9.90
# nvidia-cusolver-cu12==11.7.3.90
# nvidia-cusparse-cu12==12.5.8.93
# nvidia-cusparselt-cu12==0.7.1
# nvidia-nccl-cu12==2.27.5
# nvidia-nvjitlink-cu12==12.8.93
# nvidia-nvshmem-cu12==3.3.20
# nvidia-nvtx-cu12==12.8.90
omegaconf==2.3.0
onnxruntime==1.23.2
orjson==3.11.4
pdfminer.six==20250506
pdftext==0.6.3
polars-runtime-32==1.35.2
polars==1.35.2
pyclipper==1.3.0.post6
pydantic-settings==2.12.0
pydub==0.25.1
pypdf==6.2.0
pypdfium2==4.30.0
python-multipart==0.0.20
reportlab==4.4.4
rich==14.2.0
robust-downloader==0.0.2
ruff==0.14.5
safehttpx==0.1.7
scikit-image==0.25.2
seaborn==0.13.2
semantic-version==2.10.0
shapely==2.1.2
shellingham==1.5.4
simsimd==6.5.3
stringzilla==4.2.3
# thop==0.1.1.post2209072238
tifffile==2025.5.10
typer==0.20.0
typing-inspection==0.4.2
# ultralytics-thop==2.0.18
# ultralytics==8.3.228
```
### 1.3 修改代码
/raid_data/home/yqk/mineru-251114/MinerU/mineru/backend/pipeline/pipeline_analyze.py, line 1
添加代码
```
# 添加MLU支持
import torch_mlu.utils.gpu_migration
# 高版本镜像为
# import torch.mlu.utils.gpu_migration
os: Ubuntu 22.04.5 LTS
cpu: Hygon Hygon C86 7490
gcu: MLU590-M9D
driver: v6.2.11
docker: 28.3.0
```
## 2. 使用方法
```
export HF_ENDPOINT=https://hf-mirror.com
mineru-api --host 0.0.0.0 --port 8009
```
## 2. 环境准备
## 3. 其他
### 2.1 使用 Dockerfile 构建镜像
### 3.1 Dify插件配置问题
给Dify的MinerU插件使用时,需将Dify的.env文件中FILES_URL设置为http://{ip}:{dify的网页访问端口}。
根据网上找到的很多回答可能是要暴露5001,并将FILES_URL设置为http://{ip}:5001,并暴露5001端口,但其实设置为dify的网页访问端口即可。
### 3.2 容器启动方式
```
export MY_CONTAINER="[容器名称]"
num=`docker ps -a|grep "$MY_CONTAINER" | wc -l`
echo $num
echo $MY_CONTAINER
if [ 0 -eq $num ];then
docker run -d \
--privileged \
--pid=host \
--net=host \
--shm-size 64g \
--device /dev/cambricon_dev0 \
--device /dev/cambricon_ipcm0 \
--device /dev/cambricon_ctl \
--name $MY_CONTAINER \
-v [/path/to/your/data:/path/to/your/data] \
-v /usr/bin/cnmon:/usr/bin/cnmon \
[镜像名称] \
sleep infinity
docker exec -ti $MY_CONTAINER /bin/bash
else
docker start $MY_CONTAINER
docker exec -ti $MY_CONTAINER /bin/bash
fi
```
### 3.3 将上面的过程进行打包
准备好前面的requirements_check.txt
Dockerfile
```
# 1. 使用指定的基础镜像
FROM cambricon-base/pytorch:v25.01-torch2.5.0-torchmlu1.24.1-ubuntu22.04-py310
# 2. 设置环境变量
ENV HF_ENDPOINT=https://hf-mirror.com
# 3. 定义 venv_pip 路径以便复用
# 基础镜像中的虚拟环境路径
ARG VENV_PIP=/torch/venv3/pytorch_infer/bin/pip
# 4. 设置工作目录
WORKDIR /app
# 5. 安装 git (基础镜像可能不包含)
RUN apt-get update && apt-get install -y git && \
rm -rf /var/lib/apt/lists/*
# 6. 复制 requirements_check.txt 到镜像中
# (这个文件需要您在宿主机上和 Dockerfile 放在同一目录下)
COPY requirements_check.txt .
# 7. 步骤 1.1 & 1.2: 获取代码并安装所有依赖
# 在一个 RUN 层中执行所有安装,以优化镜像大小
RUN \
# 1.1 获取代码
echo "Cloning MinerU repository..." && \
git clone https://gh-proxy.org/https://github.com/opendatalab/MinerU.git && \
cd MinerU && \
git checkout fa1149cd4abf9db5e0f13e4e074cdb568be189f4 && \
cd .. && \
\
# 1.2 安装依赖
# 第1个pip install (来自您的步骤)
echo "Installing initial dependencies..." && \
${VENV_PIP} install accelerate==1.11.0 doclayout_yolo==0.0.4 thop==0.1.1.post2209072238 ultralytics-thop==2.0.18 ultralytics==8.3.228 && \
\
# 第2个pip install (来自 requirements_check.txt)
echo "Installing dependencies from requirements_check.txt..." && \
# 注意:基础镜像已包含 torch 和 triton,requirements_check.txt 中的注释行会被 pip 自动忽略
${VENV_PIP} install -r requirements_check.txt && \
\
# 第3个pip install (本地安装 MinerU)
echo "Installing MinerU in editable mode..." && \
cd MinerU && \
${VENV_PIP} install -e .[core] --no-deps
# 8. 步骤 1.3: 修改代码
# 将 MLU 支持代码添加到指定文件的开头
RUN echo "Applying MLU patch to pipeline_analyze.py..." && \
sed -i '1i# 添加MLU支持\nimport torch_mlu.utils.gpu_migration\n# 高版本镜像为\n# import torch.mlu.utils.gpu_migration\n' \
/app/MinerU/mineru/backend/pipeline/pipeline_analyze.py
```
该镜像的启动
```
docker run -d --restart=always \
--privileged \
--pid=host \
--net=host \
--shm-size 64g \
--device /dev/cambricon_dev0 \
--device /dev/cambricon_ipcm0 \
--device /dev/cambricon_ctl \
--name mineru_service \
mineru-mlu:latest \
/torch/venv3/pytorch_infer/bin/python /app/MinerU/mineru/cli/fast_api.py --host 0.0.0.0 --port 8009
```bash
wget https://gcore.jsdelivr.net/gh/opendatalab/MinerU@master/docker/china/mlu.Dockerfile
docker build --network=host -t mineru:mlu-lmdeploy-latest -f mlu.Dockerfile .
```
## 3. 启动 Docker 容器
```bash
docker run --name mineru_docker \
--privileged \
--ipc=host \
--network=host \
--cap-add SYS_PTRACE \
--device=/dev/mem \
--device=/dev/dri \
--device=/dev/infiniband \
--device=/dev/cambricon_ctl \
--device=/dev/cambricon_dev0 \
--device=/dev/cambricon_dev1 \
--device=/dev/cambricon_dev2 \
--device=/dev/cambricon_dev3 \
--device=/dev/cambricon_dev4 \
--device=/dev/cambricon_dev5 \
--device=/dev/cambricon_dev6 \
--device=/dev/cambricon_dev7 \
--group-add video \
--shm-size=400g \
--ulimit memlock=-1 \
--security-opt seccomp=unconfined \
--security-opt apparmor=unconfined \
-e MINERU_MODEL_SOURCE=local \
-e MINERU_LMDEPLOY_DEVICE=camb \
--entrypoint /bin/bash \
-it mineru:mlu-lmdeploy-latest
```
执行该命令后,您将进入到Docker容器的交互式终端,您可以直接在容器内运行MinerU相关命令来使用MinerU的功能。
您也可以直接通过替换`/bin/bash`为服务启动命令来启动MinerU服务,详细说明请参考[通过命令启动服务](https://opendatalab.github.io/MinerU/zh/usage/quick_usage/#apiwebuihttp-clientserver)。
## 4. 注意事项
不同环境下,MinerU对Cambricon加速卡的支持情况如下表所示:
<table border="1">
<thead>
<tr>
<th rowspan="2" colspan="2">使用场景</th>
<th colspan="2">容器环境</th>
</tr>
<tr>
<th>lmdeploy</th>
</tr>
</thead>
<tbody>
<tr>
<td rowspan="3">命令行工具(mineru)</td>
<td>pipeline</td>
<td>🟢</td>
</tr>
<tr>
<td>&lt;vlm/hybrid&gt;-auto-engine</td>
<td>🟢</td>
</tr>
<tr>
<td>&lt;vlm/hybrid&gt;-http-client</td>
<td>🟢</td>
</tr>
<tr>
<td rowspan="3">fastapi服务(mineru-api)</td>
<td>pipeline</td>
<td>🟢</td>
</tr>
<tr>
<td>&lt;vlm/hybrid&gt;-auto-engine</td>
<td>🟢</td>
</tr>
<tr>
<td>&lt;vlm/hybrid&gt;-http-client</td>
<td>🟢</td>
</tr>
<tr>
<td rowspan="3">gradio界面(mineru-gradio)</td>
<td>pipeline</td>
<td>🟢</td>
</tr>
<tr>
<td>&lt;vlm/hybrid&gt;-auto-engine</td>
<td>🟢</td>
</tr>
<tr>
<td>&lt;vlm/hybrid&gt;-http-client</td>
<td>🟢</td>
</tr>
<tr>
<td colspan="2">openai-server服务(mineru-openai-server)</td>
<td>🟢</td>
</tr>
<tr>
<td colspan="2">数据并行 (--data-parallel-size)</td>
<td>🟢</td>
</tr>
</tbody>
</table>
注:
🟢: 支持,运行较稳定,精度与Nvidia GPU基本一致
🟡: 支持但较不稳定,在某些场景下可能出现异常,或精度存在一定差异
🔴: 不支持,无法运行,或精度存在较大差异
>[!TIP]
>Cambricon加速卡指定可用加速卡的方式与NVIDIA GPU类似,请参考[使用指定GPU设备](https://opendatalab.github.io/MinerU/zh/usage/advanced_cli_parameters/#cuda_visible_devices)章节说明,
+2 -1
View File
@@ -15,9 +15,10 @@
* [海光 Hygon](acceleration_cards/Hygon.md) 🚀
* [燧原 Enflame](acceleration_cards/Enflame.md) 🚀
* [摩尔线程 MooreThreads](acceleration_cards/MooreThreads.md) 🚀
* [天数智芯 IluvatarCorex](acceleration_cards/IluvatarCorex.md) 🚀
* [寒武纪 Cambricon](acceleration_cards/Cambricon.md) 🚀
* [AMD](acceleration_cards/AMD.md) [#3662](https://github.com/opendatalab/MinerU/discussions/3662) ❤️
* [太初元碁 Tecorigin](acceleration_cards/Tecorigin.md) [#3767](https://github.com/opendatalab/MinerU/pull/3767) ❤️
* [寒武纪 Cambricon](acceleration_cards/Cambricon.md) [#4004](https://github.com/opendatalab/MinerU/discussions/4004) ❤️
* [瀚博 VastAI](acceleration_cards/VastAI.md) [#4237](https://github.com/opendatalab/MinerU/discussions/4237)❤️
- 插件与生态
* [Cherry Studio](plugin/Cherry_Studio.md)
+4
View File
@@ -198,6 +198,10 @@ def model_init(model_name: str):
if hasattr(torch, 'npu') and torch.npu.is_available():
if torch.npu.is_bf16_supported():
bf_16_support = True
elif device_name.startswith("mlu"):
if hasattr(torch, 'mlu') and torch.mlu.is_available():
if torch.mlu.is_bf16_supported():
bf_16_support = True
if model_name == 'layoutreader':
# 检测modelscope的缓存目录是否存在
+5 -1
View File
@@ -94,7 +94,11 @@ def get_device():
if torch.musa.is_available():
return "musa"
except Exception as e:
pass
try:
if torch.mlu.is_available():
return "mlu"
except Exception as e:
pass
return "cpu"
+6
View File
@@ -429,6 +429,9 @@ def clean_memory(device='cuda'):
elif str(device).startswith("musa"):
if torch.musa.is_available():
torch.musa.empty_cache()
elif str(device).startswith("mlu"):
if torch.mlu.is_available():
torch.mlu.empty_cache()
gc.collect()
@@ -470,5 +473,8 @@ def get_vram(device) -> int:
elif str(device).startswith("musa"):
if torch.musa.is_available():
total_memory = round(torch.musa.get_device_properties(device).total_memory / (1024 ** 3)) # 转为 GB
elif str(device).startswith("mlu"):
if torch.mlu.is_available():
total_memory = round(torch.mlu.get_device_properties(device).total_memory / (1024 ** 3)) # 转为 GB
return total_memory