From 6c656af65f2f423c7c41dc89fac48f7efa86d776 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E8=B5=B5=E5=B0=8F=E8=92=99?= Date: Fri, 28 Jun 2024 14:41:12 +0800 Subject: [PATCH] update:cleanup requirements.txt --- magic_pdf/libs/path_utils.py | 17 +++++++++++++---- requirements-qa.txt | 14 ++++++++++++++ requirements.txt | 17 +---------------- 3 files changed, 28 insertions(+), 20 deletions(-) create mode 100644 requirements-qa.txt diff --git a/magic_pdf/libs/path_utils.py b/magic_pdf/libs/path_utils.py index 70c77967..15fff01b 100644 --- a/magic_pdf/libs/path_utils.py +++ b/magic_pdf/libs/path_utils.py @@ -1,7 +1,5 @@ -from s3pathlib import S3Path - def remove_non_official_s3_args(s3path): """ example: s3://abc/xxxx.json?bytes=0,81350 ==> s3://abc/xxxx.json @@ -10,8 +8,19 @@ def remove_non_official_s3_args(s3path): return arr[0] def parse_s3path(s3path: str): - p = S3Path(remove_non_official_s3_args(s3path)) - return p.bucket, p.key + # from s3pathlib import S3Path + # p = S3Path(remove_non_official_s3_args(s3path)) + # return p.bucket, p.key + s3path = remove_non_official_s3_args(s3path).strip() + if s3path.startswith(('s3://', 's3a://')): + prefix, path = s3path.split('://', 1) + bucket_name, key = path.split('/', 1) + return bucket_name, key + elif s3path.startswith('/'): + raise ValueError("The provided path starts with '/'. This does not conform to a valid S3 path format.") + else: + raise ValueError("Invalid S3 path format. Expected 's3://bucket-name/key' or 's3a://bucket-name/key'.") + def parse_s3_range_params(s3path: str): """ diff --git a/requirements-qa.txt b/requirements-qa.txt new file mode 100644 index 00000000..185c25c7 --- /dev/null +++ b/requirements-qa.txt @@ -0,0 +1,14 @@ +Levenshtein +nltk +rapidfuzz +statistics +openxlab #安装opendatalab +pandas +numpy +matplotlib +seaborn +scipy +scikit-learn +tqdm +htmltabletomd +pypandoc \ No newline at end of file diff --git a/requirements.txt b/requirements.txt index de76acb0..cbd71e82 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,26 +1,11 @@ boto3>=1.28.43 Brotli>=1.1.0 click>=8.1.7 -Distance>=0.1.3 PyMuPDF>=1.24.7 loguru>=0.6.0 -matplotlib>=3.8.3 numpy>=1.21.6 -pandas>=1.3.5 fast-langdetect>=0.1.1 -regex>=2023.12.25 -termcolor>=2.4.0 wordninja>=2.0.0 scikit-learn>=1.0.2 -nltk==3.8.1 -s3pathlib>=2.1.1 pdfminer.six>=20231228 -Levenshtein -rapidfuzz -statistics -openxlab #安装opendatalab -seaborn -scipy -tqdm -htmltabletomd -pypandoc \ No newline at end of file +# requirements.txt 须保证只引入必需的外部依赖,如有新依赖添加请联系项目管理员 \ No newline at end of file