diff --git a/.gitignore b/.gitignore index f86aeca2..b6ab4538 100644 --- a/.gitignore +++ b/.gitignore @@ -1,45 +1,50 @@ -*.tar -*.tar.gz -*.zip -venv*/ -envs/ -slurm_logs/ - -sync1.sh -data_preprocess_pj1 -data-preparation1 -__pycache__ -*.log -*.pyc -.vscode -debug/ -*.ipynb -.idea - -# vscode history -.history - -.DS_Store -.env - -bad_words/ -bak/ - -app/tests/* -temp/ -tmp/ -tmp -.vscode -.vscode/ -ocr_demo -.coveragerc -/app/common/__init__.py -/magic_pdf/config/__init__.py -source.dev.env - -tmp - -projects/web/node_modules -projects/web/dist - -projects/web_demo/web_demo/static/ +*.tar +*.tar.gz +*.zip +venv*/ +envs/ +slurm_logs/ + +sync1.sh +data_preprocess_pj1 +data-preparation1 +__pycache__ +*.log +*.pyc +.vscode +debug/ +*.ipynb +.idea + +# vscode history +.history + +.DS_Store +.env + +bad_words/ +bak/ + +app/tests/* +temp/ +tmp/ +tmp +.vscode +.vscode/ +ocr_demo +.coveragerc +/app/common/__init__.py +/magic_pdf/config/__init__.py +source.dev.env + +tmp + +projects/web/node_modules +projects/web/dist + +projects/web_demo/web_demo/static/ +cli_debug/ +debug_utils/ + +# sphinx docs +_build/ diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index b86d6f8b..fc7446df 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -3,7 +3,7 @@ repos: rev: 5.0.4 hooks: - id: flake8 - args: ["--max-line-length=120", "--ignore=E131,E125,W503,W504,E203"] + args: ["--max-line-length=150", "--ignore=E131,E125,W503,W504,E203"] - repo: https://github.com/PyCQA/isort rev: 5.11.5 hooks: @@ -12,11 +12,12 @@ repos: rev: v0.32.0 hooks: - id: yapf - args: ["--style={based_on_style: google, column_limit: 120, indent_width: 4}"] + args: ["--style={based_on_style: google, column_limit: 150, indent_width: 4}"] - repo: https://github.com/codespell-project/codespell rev: v2.2.1 hooks: - id: codespell + args: ['--skip', '*.json'] - repo: https://github.com/pre-commit/pre-commit-hooks rev: v4.3.0 hooks: diff --git a/README.md b/README.md index 3e4fd746..cea8f060 100644 --- a/README.md +++ b/README.md @@ -41,6 +41,17 @@ # Changelog +- 2024/10/31 0.9.0 released. This is a major new version with extensive code refactoring, addressing numerous issues, improving performance, reducing hardware requirements, and enhancing usability: + - Refactored the sorting module code to use [layoutreader](https://github.com/ppaanngggg/layoutreader) for reading order sorting, ensuring high accuracy in various layouts. + - Refactored the paragraph concatenation module to achieve good results in cross-column, cross-page, cross-figure, and cross-table scenarios. + - Refactored the list and table of contents recognition functions, significantly improving the accuracy of list blocks and table of contents blocks, as well as the parsing of corresponding text paragraphs. + - Refactored the matching logic for figures, tables, and descriptive text, greatly enhancing the accuracy of matching captions and footnotes to figures and tables, and reducing the loss rate of descriptive text to zero. + - Added multi-language support for OCR, supporting detection and recognition of 84 languages.For the list of supported languages, see [OCR Language Support List](https://paddlepaddle.github.io/PaddleOCR/latest/en/ppocr/blog/multi_languages.html#5-support-languages-and-abbreviations). + - Added memory recycling logic and other memory optimization measures, significantly reducing memory usage. The memory requirement for enabling all acceleration features except table acceleration (layout/formula/OCR) has been reduced from 16GB to 8GB, and the memory requirement for enabling all acceleration features has been reduced from 24GB to 10GB. + - Optimized configuration file feature switches, adding an independent formula detection switch to significantly improve speed and parsing results when formula detection is not needed. + - Integrated [PDF-Extract-Kit 1.0](https://github.com/opendatalab/PDF-Extract-Kit): + - Added the self-developed `doclayout_yolo` model, which speeds up processing by more than 10 times compared to the original solution while maintaining similar parsing effects, and can be freely switched with `layoutlmv3` via the configuration file. + - Upgraded formula parsing to `unimernet 0.2.1`, improving formula parsing accuracy while significantly reducing memory usage. - 2024/09/27 Version 0.8.1 released, Fixed some bugs, and providing a [localized deployment version](projects/web_demo/README.md) of the [online demo](https://opendatalab.com/OpenSourceTools/Extractor/PDF/) and the [front-end interface](projects/web/README.md). - 2024/09/09: Version 0.8.0 released, supporting fast deployment with Dockerfile, and launching demos on Huggingface and Modelscope. - 2024/08/30: Version 0.7.1 released, add paddle tablemaster table recognition option @@ -69,6 +80,7 @@ @@ -100,15 +112,18 @@ https://github.com/user-attachments/assets/4bea02c9-6d54-4cd6-97ed-dff14340982c ## Key Features -- Removes elements such as headers, footers, footnotes, and page numbers while maintaining semantic continuity -- Outputs text in a human-readable order from multi-column documents -- Retains the original structure of the document, including titles, paragraphs, and lists -- Extracts images, image captions, tables, and table captions -- Automatically recognizes formulas in the document and converts them to LaTeX -- Automatically recognizes tables in the document and converts them to LaTeX -- Automatically detects and enables OCR for corrupted PDFs -- Supports both CPU and GPU environments -- Supports Windows, Linux, and Mac platforms +- Remove headers, footers, footnotes, page numbers, etc., to ensure semantic coherence. +- Output text in human-readable order, suitable for single-column, multi-column, and complex layouts. +- Preserve the structure of the original document, including headings, paragraphs, lists, etc. +- Extract images, image descriptions, tables, table titles, and footnotes. +- Automatically recognize and convert formulas in the document to LaTeX format. +- Automatically recognize and convert tables in the document to LaTeX or HTML format. +- Automatically detect scanned PDFs and garbled PDFs and enable OCR functionality. +- OCR supports detection and recognition of 84 languages. +- Supports multiple output formats, such as multimodal and NLP Markdown, JSON sorted by reading order, and rich intermediate formats. +- Supports various visualization results, including layout visualization and span visualization, for efficient confirmation of output quality. +- Supports both CPU and GPU environments. +- Compatible with Windows, Linux, and Mac platforms. ## Quick Start @@ -139,8 +154,8 @@ In non-mainline environments, due to the diversity of hardware and software conf CPU - x86_64 - x86_64 + x86_64(unsupported ARM Linux) + x86_64(unsupported ARM Windows) x86_64 / arm64 @@ -149,7 +164,7 @@ In non-mainline environments, due to the diversity of hardware and software conf Python Version - 3.10 + 3.10(Please make sure to create a Python 3.10 virtual environment using conda) Nvidia Driver Version @@ -166,21 +181,24 @@ In non-mainline environments, due to the diversity of hardware and software conf GPU Hardware Support List Minimum Requirement 8G+ VRAM - 3060ti/3070/3080/3080ti/4060/4070/4070ti
+ 3060ti/3070/4060
8G VRAM enables layout, formula recognition acceleration and OCR acceleration None - Recommended Configuration 16G+ VRAM - 3090/3090ti/4070ti super/4080/4090
- 16G VRAM or more can enable layout, formula recognition, OCR acceleration and table recognition acceleration simultaneously + Recommended Configuration 10G+ VRAM + 3080/3080ti/3090/3090ti/4070/4070ti/4070tisuper/4080/4090
+ 10G VRAM or more can enable layout, formula recognition, OCR acceleration and table recognition acceleration simultaneously ### Online Demo +Stable Version (Stable version verified by QA): [![OpenDataLab](https://img.shields.io/badge/Demo_on_OpenDataLab-blue?logo=data:image/svg+xml;base64,PHN2ZyB3aWR0aD0iMzAiIGhlaWdodD0iMzAiIHhtbG5zPSJodHRwOi8vd3d3LnczLm9yZy8yMDAwL3N2ZyIgZmlsbD0ibm9uZSI+CiA8ZGVmcz4KICA8bGluZWFyR3JhZGllbnQgeTI9IjAuNTMzNjciIHgyPSIxLjAwMDQiIHkxPSIwLjI5MjE5IiB4MT0iLTAuMTEyNjgiIGlkPSJhIj4KICAgPHN0b3Agc3RvcC1jb2xvcj0iIzE1NDNGRSIvPgogICA8c3RvcCBzdG9wLWNvbG9yPSIjOEM0NkZGIiBvZmZzZXQ9IjEiLz4KICA8L2xpbmVhckdyYWRpZW50PgogIDxsaW5lYXJHcmFkaWVudCB5Mj0iMC41OTc1NyIgeDI9IjEuMDExMzciIHkxPSIwLjExMDIzIiB4MT0iLTAuMDg0NzQiIGlkPSJiIj4KICAgPHN0b3Agc3RvcC1jb2xvcj0iIzE1NDNGRSIvPgogICA8c3RvcCBzdG9wLWNvbG9yPSIjOEM0NkZGIiBvZmZzZXQ9IjEiLz4KICA8L2xpbmVhckdyYWRpZW50PgogPC9kZWZzPgogPGc+CiAgPHRpdGxlPkxheWVyIDE8L3RpdGxlPgogIDxwYXRoIGlkPSJzdmdfMSIgZmlsbD0idXJsKCNhKSIgZD0ibTEuNjIzLDEyLjA2N2EwLjQ4NCwwLjQ4NCAwIDAgMSAwLjA3LC0wLjM4NGw1LjMxLC03Ljg5NWMwLjA2OCwtMC4xIDAuMTcsLTAuMTcyIDAuMjg4LC0wLjJsMTQuMzc3LC0zLjQ3NGEwLjQ4NCwwLjQ4NCAwIDAgMSAwLjU4NCwwLjM1N2wzLjY2MiwxNS4xNTJjMS40NzcsNi4xMTQgLTIuMjgxLDEyLjI2NyAtOC4zOTQsMTMuNzQ1Yy02LjExNCwxLjQ3NyAtMTIuMjY3LC0yLjI4MSAtMTMuNzQ1LC04LjM5NWwtMi4xNTIsLTguOTA2eiIgb3BhY2l0eT0iMC40Ii8+CiAgPHBhdGggaWQ9InN2Z18yIiBmaWxsPSJ1cmwoI2IpIiBkPSJtNS44MjYsOC42NzNjMCwtMC4xMzYgMC4wNTcsLTAuMjY2IDAuMTU3LC0wLjM1OGw3LjAxNywtNi40MjVhMC40ODQsMC40ODQgMCAwIDEgMC4zMjcsLTAuMTI3bDE0Ljc5LDBjMC4yNjgsMCAwLjQ4NSwwLjIxNiAwLjQ4NSwwLjQ4NGwwLDE1LjU4OWMwLDYuMjkgLTUuMDk5LDExLjM4OCAtMTEuMzg4LDExLjM4OGMtNi4yOSwwIC0xMS4zODgsLTUuMDk5IC0xMS4zODgsLTExLjM4OGwwLC05LjE2M3oiLz4KICA8cGF0aCBpZD0ic3ZnXzMiIGZpbGw9IiM1RDc2RkYiIGQ9Im0xMi4zMzEsOC43NTNsLTYuMzgzLC0wLjM5OGw3LjEyMiwtNi41MmwwLjI5OSw1Ljg5MmEwLjk3OCwwLjk3OCAwIDAgMSAtMS4wMzgsMS4wMjZ6Ii8+CiAgPHBhdGggaWQ9InN2Z180IiBmaWxsPSIjMDAyOEZEIiBkPSJtMjAuNDE2LDE1LjAyMmwwLDEuNzExYTIuNDA0LDIuNDA0IDAgMCAxIC00LjgwOCwwbDAsLTQuMjc4bC0yLjgxLDBsMCw0LjY4NmE1LjIxNSw1LjIxNSAwIDEgMCAxMC40MywwbDAsLTQuNjg2bDAsMi41NjdsLTIuODEyLDB6IiBjbGlwLXJ1bGU9ImV2ZW5vZGQiIGZpbGwtcnVsZT0iZXZlbm9kZCIvPgogIDxwYXRoIGlkPSJzdmdfNSIgZmlsbD0iIzAwMjhGRCIgZD0ibTIzLjIyOCwxMy44ODFsMS4xNCwwbDAsMS4xNDFsLTEuMTQsMGwwLC0xLjE0bDAsLTAuMDAxem0tMi44MTIsLTAuNjkybDEuODM0LDBsMCwxLjgzM2wtMS44MzQsMGwwLC0xLjgzMmwwLC0wLjAwMXptMS44MzQsLTAuOTc5bDAuOTc4LDBsMCwwLjk3OWwtMC45NzgsMGwwLC0wLjk3OGwwLC0wLjAwMXptMS41NDgsLTEuNjI5bDAuNjExLDBsMCwwLjYxMWwtMC42MTEsMGwwLC0wLjYxMXoiLz4KICA8cGF0aCBpZD0ic3ZnXzYiIGZpbGw9IiNmZmYiIGQ9Im0yMC4wODYsMTQuOTEybDAsMS43MTFhMi40MDQsMi40MDQgMCAxIDEgLTQuODA3LDBsMCwtNC4yNzhsLTIuODEyLDBsMCw0LjY4NmE1LjIxNSw1LjIxNSAwIDAgMCAxMC40MywwbDAsLTQuNjg2bDAsMi41NjdsLTIuODEsMGwtMC4wMDEsMHoiIGNsaXAtcnVsZT0iZXZlbm9kZCIgZmlsbC1ydWxlPSJldmVub2RkIi8+CiAgPHBhdGggaWQ9InN2Z183IiBmaWxsPSIjZmZmIiBkPSJtMjIuODk4LDEzLjc3MWwxLjE0LDBsMCwxLjE0MWwtMS4xNCwwbDAsLTEuMTRsMCwtMC4wMDF6bS0yLjgxMiwtMC42OTJsMS44MzQsMGwwLDEuODMzbC0xLjgzNCwwbDAsLTEuODMybDAsLTAuMDAxem0xLjgzNCwtMC45NzlsMC45NzgsMGwwLDAuOTc5bC0wLjk3OCwwbDAsLTAuOTc5em0xLjU0OCwtMS42MjlsMC42MTEsMGwwLDAuNjExbC0wLjYxLDBsMCwtMC42MWwtMC4wMDEsLTAuMDAxeiIvPgogPC9nPgo8L3N2Zz4=&labelColor=white)](https://opendatalab.com/OpenSourceTools/Extractor/PDF) + +Test Version (Synced with dev branch updates, testing new features): [![HuggingFace](https://img.shields.io/badge/Demo_on_HuggingFace-yellow.svg?logo=data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAF8AAABYCAMAAACkl9t/AAAAk1BMVEVHcEz/nQv/nQv/nQr/nQv/nQr/nQv/nQv/nQr/wRf/txT/pg7/yRr/rBD/zRz/ngv/oAz/zhz/nwv/txT/ngv/0B3+zBz/nQv/0h7/wxn/vRb/thXkuiT/rxH/pxD/ogzcqyf/nQvTlSz/czCxky7/SjifdjT/Mj3+Mj3wMj15aTnDNz+DSD9RTUBsP0FRO0Q6O0WyIxEIAAAAGHRSTlMADB8zSWF3krDDw8TJ1NbX5efv8ff9/fxKDJ9uAAAGKklEQVR42u2Z63qjOAyGC4RwCOfB2JAGqrSb2WnTw/1f3UaWcSGYNKTdf/P+mOkTrE+yJBulvfvLT2A5ruenaVHyIks33npl/6C4s/ZLAM45SOi/1FtZPyFur1OYofBX3w7d54Bxm+E8db+nDr12ttmESZ4zludJEG5S7TO72YPlKZFyE+YCYUJTBZsMiNS5Sd7NlDmKM2Eg2JQg8awbglfqgbhArjxkS7dgp2RH6hc9AMLdZYUtZN5DJr4molC8BfKrEkPKEnEVjLbgW1fLy77ZVOJagoIcLIl+IxaQZGjiX597HopF5CkaXVMDO9Pyix3AFV3kw4lQLCbHuMovz8FallbcQIJ5Ta0vks9RnolbCK84BtjKRS5uA43hYoZcOBGIG2Epbv6CvFVQ8m8loh66WNySsnN7htL58LNp+NXT8/PhXiBXPMjLSxtwp8W9f/1AngRierBkA+kk/IpUSOeKByzn8y3kAAAfh//0oXgV4roHm/kz4E2z//zRc3/lgwBzbM2mJxQEa5pqgX7d1L0htrhx7LKxOZlKbwcAWyEOWqYSI8YPtgDQVjpB5nvaHaSnBaQSD6hweDi8PosxD6/PT09YY3xQA7LTCTKfYX+QHpA0GCcqmEHvr/cyfKQTEuwgbs2kPxJEB0iNjfJcCTPyocx+A0griHSmADiC91oNGVwJ69RudYe65vJmoqfpul0lrqXadW0jFKH5BKwAeCq+Den7s+3zfRJzA61/Uj/9H/VzLKTx9jFPPdXeeP+L7WEvDLAKAIoF8bPTKT0+TM7W8ePj3Rz/Yn3kOAp2f1Kf0Weony7pn/cPydvhQYV+eFOfmOu7VB/ViPe34/EN3RFHY/yRuT8ddCtMPH/McBAT5s+vRde/gf2c/sPsjLK+m5IBQF5tO+h2tTlBGnP6693JdsvofjOPnnEHkh2TnV/X1fBl9S5zrwuwF8NFrAVJVwCAPTe8gaJlomqlp0pv4Pjn98tJ/t/fL++6unpR1YGC2n/KCoa0tTLoKiEeUPDl94nj+5/Tv3/eT5vBQ60X1S0oZr+IWRR8Ldhu7AlLjPISlJcO9vrFotky9SpzDequlwEir5beYAc0R7D9KS1DXva0jhYRDXoExPdc6yw5GShkZXe9QdO/uOvHofxjrV/TNS6iMJS+4TcSTgk9n5agJdBQbB//IfF/HpvPt3Tbi7b6I6K0R72p6ajryEJrENW2bbeVUGjfgoals4L443c7BEE4mJO2SpbRngxQrAKRudRzGQ8jVOL2qDVjjI8K1gc3TIJ5KiFZ1q+gdsARPB4NQS4AjwVSt72DSoXNyOWUrU5mQ9nRYyjp89Xo7oRI6Bga9QNT1mQ/ptaJq5T/7WcgAZywR/XlPGAUDdet3LE+qS0TI+g+aJU8MIqjo0Kx8Ly+maxLjJmjQ18rA0YCkxLQbUZP1WqdmyQGJLUm7VnQFqodmXSqmRrdVpqdzk5LvmvgtEcW8PMGdaS23EOWyDVbACZzUJPaqMbjDxpA3Qrgl0AikimGDbqmyT8P8NOYiqrldF8rX+YN7TopX4UoHuSCYY7cgX4gHwclQKl1zhx0THf+tCAUValzjI7Wg9EhptrkIcfIJjA94evOn8B2eHaVzvBrnl2ig0So6hvPaz0IGcOvTHvUIlE2+prqAxLSQxZlU2stql1NqCCLdIiIN/i1DBEHUoElM9dBravbiAnKqgpi4IBkw+utSPIoBijDXJipSVV7MpOEJUAc5Qmm3BnUN+w3hteEieYKfRZSIUcXKMVf0u5wD4EwsUNVvZOtUT7A2GkffHjByWpHqvRBYrTV72a6j8zZ6W0DTE86Hn04bmyWX3Ri9WH7ZU6Q7h+ZHo0nHUAcsQvVhXRDZHChwiyi/hnPuOsSEF6Exk3o6Y9DT1eZ+6cASXk2Y9k+6EOQMDGm6WBK10wOQJCBwren86cPPWUcRAnTVjGcU1LBgs9FURiX/e6479yZcLwCBmTxiawEwrOcleuu12t3tbLv/N4RLYIBhYexm7Fcn4OJcn0+zc+s8/VfPeddZHAGN6TT8eGczHdR/Gts1/MzDkThr23zqrVfAMFT33Nx1RJsx1k5zuWILLnG/vsH+Fv5D4NTVcp1Gzo8AAAAAElFTkSuQmCC&labelColor=white)](https://huggingface.co/spaces/opendatalab/MinerU) [![ModelScope](https://img.shields.io/badge/Demo_on_ModelScope-purple?logo=data:image/svg+xml;base64,PHN2ZyB3aWR0aD0iMjIzIiBoZWlnaHQ9IjIwMCIgeG1sbnM9Imh0dHA6Ly93d3cudzMub3JnLzIwMDAvc3ZnIj4KCiA8Zz4KICA8dGl0bGU+TGF5ZXIgMTwvdGl0bGU+CiAgPHBhdGggaWQ9InN2Z18xNCIgZmlsbD0iIzYyNGFmZiIgZD0ibTAsODkuODRsMjUuNjUsMGwwLDI1LjY0OTk5bC0yNS42NSwwbDAsLTI1LjY0OTk5eiIvPgogIDxwYXRoIGlkPSJzdmdfMTUiIGZpbGw9IiM2MjRhZmYiIGQ9Im05OS4xNCwxMTUuNDlsMjUuNjUsMGwwLDI1LjY1bC0yNS42NSwwbDAsLTI1LjY1eiIvPgogIDxwYXRoIGlkPSJzdmdfMTYiIGZpbGw9IiM2MjRhZmYiIGQ9Im0xNzYuMDksMTQxLjE0bC0yNS42NDk5OSwwbDAsMjIuMTlsNDcuODQsMGwwLC00Ny44NGwtMjIuMTksMGwwLDI1LjY1eiIvPgogIDxwYXRoIGlkPSJzdmdfMTciIGZpbGw9IiMzNmNmZDEiIGQ9Im0xMjQuNzksODkuODRsMjUuNjUsMGwwLDI1LjY0OTk5bC0yNS42NSwwbDAsLTI1LjY0OTk5eiIvPgogIDxwYXRoIGlkPSJzdmdfMTgiIGZpbGw9IiMzNmNmZDEiIGQ9Im0wLDY0LjE5bDI1LjY1LDBsMCwyNS42NWwtMjUuNjUsMGwwLC0yNS42NXoiLz4KICA8cGF0aCBpZD0ic3ZnXzE5IiBmaWxsPSIjNjI0YWZmIiBkPSJtMTk4LjI4LDg5Ljg0bDI1LjY0OTk5LDBsMCwyNS42NDk5OWwtMjUuNjQ5OTksMGwwLC0yNS42NDk5OXoiLz4KICA8cGF0aCBpZD0ic3ZnXzIwIiBmaWxsPSIjMzZjZmQxIiBkPSJtMTk4LjI4LDY0LjE5bDI1LjY0OTk5LDBsMCwyNS42NWwtMjUuNjQ5OTksMGwwLC0yNS42NXoiLz4KICA8cGF0aCBpZD0ic3ZnXzIxIiBmaWxsPSIjNjI0YWZmIiBkPSJtMTUwLjQ0LDQybDAsMjIuMTlsMjUuNjQ5OTksMGwwLDI1LjY1bDIyLjE5LDBsMCwtNDcuODRsLTQ3Ljg0LDB6Ii8+CiAgPHBhdGggaWQ9InN2Z18yMiIgZmlsbD0iIzM2Y2ZkMSIgZD0ibTczLjQ5LDg5Ljg0bDI1LjY1LDBsMCwyNS42NDk5OWwtMjUuNjUsMGwwLC0yNS42NDk5OXoiLz4KICA8cGF0aCBpZD0ic3ZnXzIzIiBmaWxsPSIjNjI0YWZmIiBkPSJtNDcuODQsNjQuMTlsMjUuNjUsMGwwLC0yMi4xOWwtNDcuODQsMGwwLDQ3Ljg0bDIyLjE5LDBsMCwtMjUuNjV6Ii8+CiAgPHBhdGggaWQ9InN2Z18yNCIgZmlsbD0iIzYyNGFmZiIgZD0ibTQ3Ljg0LDExNS40OWwtMjIuMTksMGwwLDQ3Ljg0bDQ3Ljg0LDBsMCwtMjIuMTlsLTI1LjY1LDBsMCwtMjUuNjV6Ii8+CiA8L2c+Cjwvc3ZnPg==&labelColor=white)](https://www.modelscope.cn/studios/OpenDataLab/MinerU) @@ -211,10 +229,18 @@ You can modify certain configurations in this file to enable or disable features ```json { - // other config - "table-config": { - "model": "TableMaster", // Another option of this value is 'struct_eqtable' - "is_table_recog_enable": false, // Table recognition is disabled by default, modify this value to enable it + // other config + "layout-config": { + "model": "layoutlmv3" // Please change to "doclayout_yolo" when using doclayout_yolo. + }, + "formula-config": { + "mfd_model": "yolo_v8_mfd", + "mfr_model": "unimernet_small", + "enable": true // The formula recognition feature is enabled by default. If you need to disable it, please change the value here to "false". + }, + "table-config": { + "model": "tablemaster", // When using structEqTable, please change to "struct_eqtable". + "enable": false, // The table recognition feature is disabled by default. If you need to enable it, please change the value here to "true". "max_time": 400 } } @@ -263,8 +289,8 @@ Options: -l, --lang TEXT Input the languages in the pdf (if known) to improve OCR accuracy. Optional. You should input "Abbreviation" with language form url: ht - tps://paddlepaddle.github.io/PaddleOCR/en/ppocr - /blog/multi_languages.html#5-support-languages- + tps://paddlepaddle.github.io/PaddleOCR/latest/en + /ppocr/blog/multi_languages.html#5-support-languages- and-abbreviations -d, --debug BOOLEAN Enables detailed debugging information during the execution of the CLI commands. @@ -288,7 +314,7 @@ The results will be saved in the `{some_output_dir}` directory. The output file ```text ├── some_pdf.md # markdown file ├── images # directory for storing images -├── some_pdf_layout.pdf # layout diagram +├── some_pdf_layout.pdf # layout diagram (Include layout reading order) ├── some_pdf_middle.json # MinerU intermediate processing result ├── some_pdf_model.json # model inference result ├── some_pdf_origin.pdf # original PDF file @@ -333,29 +359,38 @@ For detailed implementation, refer to: - [demo.py Simplest Processing Method](demo/demo.py) - [magic_pdf_parse_main.py More Detailed Processing Workflow](demo/magic_pdf_parse_main.py) +### Deploy Derived Projects + +Derived projects include secondary development projects based on MinerU by project developers and community developers, +such as application interfaces based on Gradio, RAG based on llama, web demos similar to the official website, lightweight multi-GPU load balancing client/server ends, etc. +These projects may offer more features and a better user experience. +For specific deployment methods, please refer to the [Derived Project README](projects/README.md) + + ### Development Guide TODO # TODO -- [x] Semantic-based reading order -- [ ] List recognition within the text -- [ ] Code block recognition within the text -- [ ] Table of contents recognition -- [x] Table recognition -- [ ] [Chemical formula recognition](docs/chemical_knowledge_introduction/introduction.pdf) -- [ ] Geometric shape recognition +- 🗹 Reading order based on the model +- 🗹 Recognition of `index` and `list` in the main text +- 🗹 Table recognition +- ☐ Code block recognition in the main text +- ☐ [Chemical formula recognition](docs/chemical_knowledge_introduction/introduction.pdf) +- ☐ Geometric shape recognition # Known Issues -- Reading order is segmented based on rules, which can cause disordered sequences in some cases -- Vertical text is not supported -- Lists, code blocks, and table of contents are not yet supported in the layout model -- Comic books, art books, elementary school textbooks, and exercise books are not well-parsed yet -- Enabling OCR may produce better results in PDFs with a high density of formulas -- If you are processing PDFs with a large number of formulas, it is strongly recommended to enable the OCR function. When using PyMuPDF to extract text, overlapping text lines can occur, leading to inaccurate formula insertion positions. - +- Reading order is determined by the model based on the spatial distribution of readable content, and may be out of order in some areas under extremely complex layouts. +- Vertical text is not supported. +- Tables of contents and lists are recognized through rules, and some uncommon list formats may not be recognized. +- Only one level of headings is supported; hierarchical headings are not currently supported. +- Code blocks are not yet supported in the layout model. +- Comic books, art albums, primary school textbooks, and exercises cannot be parsed well. +- Table recognition may result in row/column recognition errors in complex tables. +- OCR recognition may produce inaccurate characters in PDFs of lesser-known languages (e.g., diacritical marks in Latin script, easily confused characters in Arabic script). +- Some formulas may not render correctly in Markdown. # FAQ diff --git a/README_zh-CN.md b/README_zh-CN.md index 3eef92e4..bf6e802b 100644 --- a/README_zh-CN.md +++ b/README_zh-CN.md @@ -41,6 +41,18 @@ # 更新记录 + +- 2024/10/31 0.9.0发布,这是我们进行了大量代码重构的全新版本,解决了众多问题,提升了性能,降低了硬件需求,并提供了更丰富的易用性: + - 重构排序模块代码,使用 [layoutreader](https://github.com/ppaanngggg/layoutreader) 进行阅读顺序排序,确保在各种排版下都能实现极高准确率 + - 重构段落拼接模块,在跨栏、跨页、跨图、跨表情况下均能实现良好的段落拼接效果 + - 重构列表和目录识别功能,极大提升列表块和目录块识别的准确率及对应文本段落的解析效果 + - 重构图、表与描述性文本的匹配逻辑,大幅提升 caption 和 footnote 与图表的匹配准确率,并将描述性文本的丢失率降至零 + - 增加 OCR 的多语言支持,支持 84 种语言的检测与识别,语言支持列表详见 [OCR 语言支持列表](https://paddlepaddle.github.io/PaddleOCR/latest/ppocr/blog/multi_languages.html#5) + - 增加显存回收逻辑及其他显存优化措施,大幅降低显存使用需求。开启除表格加速外的全部加速功能(layout/公式/OCR)的显存需求从16GB降至8GB,开启全部加速功能的显存需求从24GB降至10GB + - 优化配置文件的功能开关,增加独立的公式检测开关,无需公式检测时可大幅提升速度和解析效果 + - 集成 [PDF-Extract-Kit 1.0](https://github.com/opendatalab/PDF-Extract-Kit) + - 加入自研的 `doclayout_yolo` 模型,在相近解析效果情况下比原方案提速10倍以上,可通过配置文件与 `layoutlmv3` 自由切换 + - 公式解析升级至 `unimernet 0.2.1`,在提升公式解析准确率的同时,大幅降低显存需求 - 2024/09/27 0.8.1发布,修复了一些bug,同时提供了[在线demo](https://opendatalab.com/OpenSourceTools/Extractor/PDF/)的[本地化部署版本](projects/web_demo/README_zh-CN.md)和[前端界面](projects/web/README_zh-CN.md) - 2024/09/09 0.8.0发布,支持Dockerfile快速部署,同时上线了huggingface、modelscope demo - 2024/08/30 0.7.1发布,集成了paddle tablemaster表格识别功能 @@ -69,6 +81,7 @@ @@ -100,15 +113,18 @@ https://github.com/user-attachments/assets/4bea02c9-6d54-4cd6-97ed-dff14340982c ## 主要功能 -- 删除页眉、页脚、脚注、页码等元素,保持语义连贯 -- 对多栏输出符合人类阅读顺序的文本 +- 删除页眉、页脚、脚注、页码等元素,确保语义连贯 +- 输出符合人类阅读顺序的文本,适用于单栏、多栏及复杂排版 - 保留原文档的结构,包括标题、段落、列表等 -- 提取图像、图片标题、表格、表格标题 -- 自动识别文档中的公式并将公式转换成latex -- 自动识别文档中的表格并将表格转换成latex -- 乱码PDF自动检测并启用OCR +- 提取图像、图片描述、表格、表格标题及脚注 +- 自动识别并转换文档中的公式为LaTeX格式 +- 自动识别并转换文档中的表格为LaTeX或HTML格式 +- 自动检测扫描版PDF和乱码PDF,并启用OCR功能 +- OCR支持84种语言的检测与识别 +- 支持多种输出格式,如多模态与NLP的Markdown、按阅读顺序排序的JSON、含有丰富信息的中间格式等 +- 支持多种可视化结果,包括layout可视化、span可视化等,便于高效确认输出效果与质检 - 支持CPU和GPU环境 -- 支持windows/linux/mac平台 +- 兼容Windows、Linux和Mac平台 ## 快速开始 @@ -139,8 +155,8 @@ https://github.com/user-attachments/assets/4bea02c9-6d54-4cd6-97ed-dff14340982c CPU - x86_64 - x86_64 + x86_64(暂不支持ARM Linux) + x86_64(暂不支持ARM Windows) x86_64 / arm64 @@ -149,7 +165,7 @@ https://github.com/user-attachments/assets/4bea02c9-6d54-4cd6-97ed-dff14340982c python版本 - 3.10 + 3.10 (请务必通过conda创建3.10虚拟环境) Nvidia Driver 版本 @@ -166,23 +182,27 @@ https://github.com/user-attachments/assets/4bea02c9-6d54-4cd6-97ed-dff14340982c GPU硬件支持列表 最低要求 8G+显存 - 3060ti/3070/3080/3080ti/4060/4070/4070ti
+ 3060ti/3070/4060
8G显存可开启layout、公式识别和ocr加速 None - 推荐配置 16G+显存 - 3090/3090ti/4070tisuper/4080/4090
- 16G显存及以上可以同时开启layout、公式识别和ocr加速和表格识别加速
+ 推荐配置 10G+显存 + 3080/3080ti/3090/3090ti/4070/4070ti/4070tisuper/4080/4090
+ 10G显存及以上可以同时开启layout、公式识别和ocr加速和表格识别加速
### 在线体验 +稳定版(经过QA验证的稳定版本): [![OpenDataLab](https://img.shields.io/badge/Demo_on_OpenDataLab-blue?logo=data:image/svg+xml;base64,PHN2ZyB3aWR0aD0iMzAiIGhlaWdodD0iMzAiIHhtbG5zPSJodHRwOi8vd3d3LnczLm9yZy8yMDAwL3N2ZyIgZmlsbD0ibm9uZSI+CiA8ZGVmcz4KICA8bGluZWFyR3JhZGllbnQgeTI9IjAuNTMzNjciIHgyPSIxLjAwMDQiIHkxPSIwLjI5MjE5IiB4MT0iLTAuMTEyNjgiIGlkPSJhIj4KICAgPHN0b3Agc3RvcC1jb2xvcj0iIzE1NDNGRSIvPgogICA8c3RvcCBzdG9wLWNvbG9yPSIjOEM0NkZGIiBvZmZzZXQ9IjEiLz4KICA8L2xpbmVhckdyYWRpZW50PgogIDxsaW5lYXJHcmFkaWVudCB5Mj0iMC41OTc1NyIgeDI9IjEuMDExMzciIHkxPSIwLjExMDIzIiB4MT0iLTAuMDg0NzQiIGlkPSJiIj4KICAgPHN0b3Agc3RvcC1jb2xvcj0iIzE1NDNGRSIvPgogICA8c3RvcCBzdG9wLWNvbG9yPSIjOEM0NkZGIiBvZmZzZXQ9IjEiLz4KICA8L2xpbmVhckdyYWRpZW50PgogPC9kZWZzPgogPGc+CiAgPHRpdGxlPkxheWVyIDE8L3RpdGxlPgogIDxwYXRoIGlkPSJzdmdfMSIgZmlsbD0idXJsKCNhKSIgZD0ibTEuNjIzLDEyLjA2N2EwLjQ4NCwwLjQ4NCAwIDAgMSAwLjA3LC0wLjM4NGw1LjMxLC03Ljg5NWMwLjA2OCwtMC4xIDAuMTcsLTAuMTcyIDAuMjg4LC0wLjJsMTQuMzc3LC0zLjQ3NGEwLjQ4NCwwLjQ4NCAwIDAgMSAwLjU4NCwwLjM1N2wzLjY2MiwxNS4xNTJjMS40NzcsNi4xMTQgLTIuMjgxLDEyLjI2NyAtOC4zOTQsMTMuNzQ1Yy02LjExNCwxLjQ3NyAtMTIuMjY3LC0yLjI4MSAtMTMuNzQ1LC04LjM5NWwtMi4xNTIsLTguOTA2eiIgb3BhY2l0eT0iMC40Ii8+CiAgPHBhdGggaWQ9InN2Z18yIiBmaWxsPSJ1cmwoI2IpIiBkPSJtNS44MjYsOC42NzNjMCwtMC4xMzYgMC4wNTcsLTAuMjY2IDAuMTU3LC0wLjM1OGw3LjAxNywtNi40MjVhMC40ODQsMC40ODQgMCAwIDEgMC4zMjcsLTAuMTI3bDE0Ljc5LDBjMC4yNjgsMCAwLjQ4NSwwLjIxNiAwLjQ4NSwwLjQ4NGwwLDE1LjU4OWMwLDYuMjkgLTUuMDk5LDExLjM4OCAtMTEuMzg4LDExLjM4OGMtNi4yOSwwIC0xMS4zODgsLTUuMDk5IC0xMS4zODgsLTExLjM4OGwwLC05LjE2M3oiLz4KICA8cGF0aCBpZD0ic3ZnXzMiIGZpbGw9IiM1RDc2RkYiIGQ9Im0xMi4zMzEsOC43NTNsLTYuMzgzLC0wLjM5OGw3LjEyMiwtNi41MmwwLjI5OSw1Ljg5MmEwLjk3OCwwLjk3OCAwIDAgMSAtMS4wMzgsMS4wMjZ6Ii8+CiAgPHBhdGggaWQ9InN2Z180IiBmaWxsPSIjMDAyOEZEIiBkPSJtMjAuNDE2LDE1LjAyMmwwLDEuNzExYTIuNDA0LDIuNDA0IDAgMCAxIC00LjgwOCwwbDAsLTQuMjc4bC0yLjgxLDBsMCw0LjY4NmE1LjIxNSw1LjIxNSAwIDEgMCAxMC40MywwbDAsLTQuNjg2bDAsMi41NjdsLTIuODEyLDB6IiBjbGlwLXJ1bGU9ImV2ZW5vZGQiIGZpbGwtcnVsZT0iZXZlbm9kZCIvPgogIDxwYXRoIGlkPSJzdmdfNSIgZmlsbD0iIzAwMjhGRCIgZD0ibTIzLjIyOCwxMy44ODFsMS4xNCwwbDAsMS4xNDFsLTEuMTQsMGwwLC0xLjE0bDAsLTAuMDAxem0tMi44MTIsLTAuNjkybDEuODM0LDBsMCwxLjgzM2wtMS44MzQsMGwwLC0xLjgzMmwwLC0wLjAwMXptMS44MzQsLTAuOTc5bDAuOTc4LDBsMCwwLjk3OWwtMC45NzgsMGwwLC0wLjk3OGwwLC0wLjAwMXptMS41NDgsLTEuNjI5bDAuNjExLDBsMCwwLjYxMWwtMC42MTEsMGwwLC0wLjYxMXoiLz4KICA8cGF0aCBpZD0ic3ZnXzYiIGZpbGw9IiNmZmYiIGQ9Im0yMC4wODYsMTQuOTEybDAsMS43MTFhMi40MDQsMi40MDQgMCAxIDEgLTQuODA3LDBsMCwtNC4yNzhsLTIuODEyLDBsMCw0LjY4NmE1LjIxNSw1LjIxNSAwIDAgMCAxMC40MywwbDAsLTQuNjg2bDAsMi41NjdsLTIuODEsMGwtMC4wMDEsMHoiIGNsaXAtcnVsZT0iZXZlbm9kZCIgZmlsbC1ydWxlPSJldmVub2RkIi8+CiAgPHBhdGggaWQ9InN2Z183IiBmaWxsPSIjZmZmIiBkPSJtMjIuODk4LDEzLjc3MWwxLjE0LDBsMCwxLjE0MWwtMS4xNCwwbDAsLTEuMTRsMCwtMC4wMDF6bS0yLjgxMiwtMC42OTJsMS44MzQsMGwwLDEuODMzbC0xLjgzNCwwbDAsLTEuODMybDAsLTAuMDAxem0xLjgzNCwtMC45NzlsMC45NzgsMGwwLDAuOTc5bC0wLjk3OCwwbDAsLTAuOTc5em0xLjU0OCwtMS42MjlsMC42MTEsMGwwLDAuNjExbC0wLjYxLDBsMCwtMC42MWwtMC4wMDEsLTAuMDAxeiIvPgogPC9nPgo8L3N2Zz4=&labelColor=white)](https://opendatalab.com/OpenSourceTools/Extractor/PDF) -[![ModelScope](https://img.shields.io/badge/Demo_on_ModelScope-purple?logo=data:image/svg+xml;base64,PHN2ZyB3aWR0aD0iMjIzIiBoZWlnaHQ9IjIwMCIgeG1sbnM9Imh0dHA6Ly93d3cudzMub3JnLzIwMDAvc3ZnIj4KCiA8Zz4KICA8dGl0bGU+TGF5ZXIgMTwvdGl0bGU+CiAgPHBhdGggaWQ9InN2Z18xNCIgZmlsbD0iIzYyNGFmZiIgZD0ibTAsODkuODRsMjUuNjUsMGwwLDI1LjY0OTk5bC0yNS42NSwwbDAsLTI1LjY0OTk5eiIvPgogIDxwYXRoIGlkPSJzdmdfMTUiIGZpbGw9IiM2MjRhZmYiIGQ9Im05OS4xNCwxMTUuNDlsMjUuNjUsMGwwLDI1LjY1bC0yNS42NSwwbDAsLTI1LjY1eiIvPgogIDxwYXRoIGlkPSJzdmdfMTYiIGZpbGw9IiM2MjRhZmYiIGQ9Im0xNzYuMDksMTQxLjE0bC0yNS42NDk5OSwwbDAsMjIuMTlsNDcuODQsMGwwLC00Ny44NGwtMjIuMTksMGwwLDI1LjY1eiIvPgogIDxwYXRoIGlkPSJzdmdfMTciIGZpbGw9IiMzNmNmZDEiIGQ9Im0xMjQuNzksODkuODRsMjUuNjUsMGwwLDI1LjY0OTk5bC0yNS42NSwwbDAsLTI1LjY0OTk5eiIvPgogIDxwYXRoIGlkPSJzdmdfMTgiIGZpbGw9IiMzNmNmZDEiIGQ9Im0wLDY0LjE5bDI1LjY1LDBsMCwyNS42NWwtMjUuNjUsMGwwLC0yNS42NXoiLz4KICA8cGF0aCBpZD0ic3ZnXzE5IiBmaWxsPSIjNjI0YWZmIiBkPSJtMTk4LjI4LDg5Ljg0bDI1LjY0OTk5LDBsMCwyNS42NDk5OWwtMjUuNjQ5OTksMGwwLC0yNS42NDk5OXoiLz4KICA8cGF0aCBpZD0ic3ZnXzIwIiBmaWxsPSIjMzZjZmQxIiBkPSJtMTk4LjI4LDY0LjE5bDI1LjY0OTk5LDBsMCwyNS42NWwtMjUuNjQ5OTksMGwwLC0yNS42NXoiLz4KICA8cGF0aCBpZD0ic3ZnXzIxIiBmaWxsPSIjNjI0YWZmIiBkPSJtMTUwLjQ0LDQybDAsMjIuMTlsMjUuNjQ5OTksMGwwLDI1LjY1bDIyLjE5LDBsMCwtNDcuODRsLTQ3Ljg0LDB6Ii8+CiAgPHBhdGggaWQ9InN2Z18yMiIgZmlsbD0iIzM2Y2ZkMSIgZD0ibTczLjQ5LDg5Ljg0bDI1LjY1LDBsMCwyNS42NDk5OWwtMjUuNjUsMGwwLC0yNS42NDk5OXoiLz4KICA8cGF0aCBpZD0ic3ZnXzIzIiBmaWxsPSIjNjI0YWZmIiBkPSJtNDcuODQsNjQuMTlsMjUuNjUsMGwwLC0yMi4xOWwtNDcuODQsMGwwLDQ3Ljg0bDIyLjE5LDBsMCwtMjUuNjV6Ii8+CiAgPHBhdGggaWQ9InN2Z18yNCIgZmlsbD0iIzYyNGFmZiIgZD0ibTQ3Ljg0LDExNS40OWwtMjIuMTksMGwwLDQ3Ljg0bDQ3Ljg0LDBsMCwtMjIuMTlsLTI1LjY1LDBsMCwtMjUuNjV6Ii8+CiA8L2c+Cjwvc3ZnPg==&labelColor=white)](https://www.modelscope.cn/studios/OpenDataLab/MinerU) + +测试版(同步dev分支更新,测试新特性): + [![HuggingFace](https://img.shields.io/badge/Demo_on_HuggingFace-yellow.svg?logo=data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAF8AAABYCAMAAACkl9t/AAAAk1BMVEVHcEz/nQv/nQv/nQr/nQv/nQr/nQv/nQv/nQr/wRf/txT/pg7/yRr/rBD/zRz/ngv/oAz/zhz/nwv/txT/ngv/0B3+zBz/nQv/0h7/wxn/vRb/thXkuiT/rxH/pxD/ogzcqyf/nQvTlSz/czCxky7/SjifdjT/Mj3+Mj3wMj15aTnDNz+DSD9RTUBsP0FRO0Q6O0WyIxEIAAAAGHRSTlMADB8zSWF3krDDw8TJ1NbX5efv8ff9/fxKDJ9uAAAGKklEQVR42u2Z63qjOAyGC4RwCOfB2JAGqrSb2WnTw/1f3UaWcSGYNKTdf/P+mOkTrE+yJBulvfvLT2A5ruenaVHyIks33npl/6C4s/ZLAM45SOi/1FtZPyFur1OYofBX3w7d54Bxm+E8db+nDr12ttmESZ4zludJEG5S7TO72YPlKZFyE+YCYUJTBZsMiNS5Sd7NlDmKM2Eg2JQg8awbglfqgbhArjxkS7dgp2RH6hc9AMLdZYUtZN5DJr4molC8BfKrEkPKEnEVjLbgW1fLy77ZVOJagoIcLIl+IxaQZGjiX597HopF5CkaXVMDO9Pyix3AFV3kw4lQLCbHuMovz8FallbcQIJ5Ta0vks9RnolbCK84BtjKRS5uA43hYoZcOBGIG2Epbv6CvFVQ8m8loh66WNySsnN7htL58LNp+NXT8/PhXiBXPMjLSxtwp8W9f/1AngRierBkA+kk/IpUSOeKByzn8y3kAAAfh//0oXgV4roHm/kz4E2z//zRc3/lgwBzbM2mJxQEa5pqgX7d1L0htrhx7LKxOZlKbwcAWyEOWqYSI8YPtgDQVjpB5nvaHaSnBaQSD6hweDi8PosxD6/PT09YY3xQA7LTCTKfYX+QHpA0GCcqmEHvr/cyfKQTEuwgbs2kPxJEB0iNjfJcCTPyocx+A0griHSmADiC91oNGVwJ69RudYe65vJmoqfpul0lrqXadW0jFKH5BKwAeCq+Den7s+3zfRJzA61/Uj/9H/VzLKTx9jFPPdXeeP+L7WEvDLAKAIoF8bPTKT0+TM7W8ePj3Rz/Yn3kOAp2f1Kf0Weony7pn/cPydvhQYV+eFOfmOu7VB/ViPe34/EN3RFHY/yRuT8ddCtMPH/McBAT5s+vRde/gf2c/sPsjLK+m5IBQF5tO+h2tTlBGnP6693JdsvofjOPnnEHkh2TnV/X1fBl9S5zrwuwF8NFrAVJVwCAPTe8gaJlomqlp0pv4Pjn98tJ/t/fL++6unpR1YGC2n/KCoa0tTLoKiEeUPDl94nj+5/Tv3/eT5vBQ60X1S0oZr+IWRR8Ldhu7AlLjPISlJcO9vrFotky9SpzDequlwEir5beYAc0R7D9KS1DXva0jhYRDXoExPdc6yw5GShkZXe9QdO/uOvHofxjrV/TNS6iMJS+4TcSTgk9n5agJdBQbB//IfF/HpvPt3Tbi7b6I6K0R72p6ajryEJrENW2bbeVUGjfgoals4L443c7BEE4mJO2SpbRngxQrAKRudRzGQ8jVOL2qDVjjI8K1gc3TIJ5KiFZ1q+gdsARPB4NQS4AjwVSt72DSoXNyOWUrU5mQ9nRYyjp89Xo7oRI6Bga9QNT1mQ/ptaJq5T/7WcgAZywR/XlPGAUDdet3LE+qS0TI+g+aJU8MIqjo0Kx8Ly+maxLjJmjQ18rA0YCkxLQbUZP1WqdmyQGJLUm7VnQFqodmXSqmRrdVpqdzk5LvmvgtEcW8PMGdaS23EOWyDVbACZzUJPaqMbjDxpA3Qrgl0AikimGDbqmyT8P8NOYiqrldF8rX+YN7TopX4UoHuSCYY7cgX4gHwclQKl1zhx0THf+tCAUValzjI7Wg9EhptrkIcfIJjA94evOn8B2eHaVzvBrnl2ig0So6hvPaz0IGcOvTHvUIlE2+prqAxLSQxZlU2stql1NqCCLdIiIN/i1DBEHUoElM9dBravbiAnKqgpi4IBkw+utSPIoBijDXJipSVV7MpOEJUAc5Qmm3BnUN+w3hteEieYKfRZSIUcXKMVf0u5wD4EwsUNVvZOtUT7A2GkffHjByWpHqvRBYrTV72a6j8zZ6W0DTE86Hn04bmyWX3Ri9WH7ZU6Q7h+ZHo0nHUAcsQvVhXRDZHChwiyi/hnPuOsSEF6Exk3o6Y9DT1eZ+6cASXk2Y9k+6EOQMDGm6WBK10wOQJCBwren86cPPWUcRAnTVjGcU1LBgs9FURiX/e6479yZcLwCBmTxiawEwrOcleuu12t3tbLv/N4RLYIBhYexm7Fcn4OJcn0+zc+s8/VfPeddZHAGN6TT8eGczHdR/Gts1/MzDkThr23zqrVfAMFT33Nx1RJsx1k5zuWILLnG/vsH+Fv5D4NTVcp1Gzo8AAAAAElFTkSuQmCC&labelColor=white)](https://huggingface.co/spaces/opendatalab/MinerU) +[![ModelScope](https://img.shields.io/badge/Demo_on_ModelScope-purple?logo=data:image/svg+xml;base64,PHN2ZyB3aWR0aD0iMjIzIiBoZWlnaHQ9IjIwMCIgeG1sbnM9Imh0dHA6Ly93d3cudzMub3JnLzIwMDAvc3ZnIj4KCiA8Zz4KICA8dGl0bGU+TGF5ZXIgMTwvdGl0bGU+CiAgPHBhdGggaWQ9InN2Z18xNCIgZmlsbD0iIzYyNGFmZiIgZD0ibTAsODkuODRsMjUuNjUsMGwwLDI1LjY0OTk5bC0yNS42NSwwbDAsLTI1LjY0OTk5eiIvPgogIDxwYXRoIGlkPSJzdmdfMTUiIGZpbGw9IiM2MjRhZmYiIGQ9Im05OS4xNCwxMTUuNDlsMjUuNjUsMGwwLDI1LjY1bC0yNS42NSwwbDAsLTI1LjY1eiIvPgogIDxwYXRoIGlkPSJzdmdfMTYiIGZpbGw9IiM2MjRhZmYiIGQ9Im0xNzYuMDksMTQxLjE0bC0yNS42NDk5OSwwbDAsMjIuMTlsNDcuODQsMGwwLC00Ny44NGwtMjIuMTksMGwwLDI1LjY1eiIvPgogIDxwYXRoIGlkPSJzdmdfMTciIGZpbGw9IiMzNmNmZDEiIGQ9Im0xMjQuNzksODkuODRsMjUuNjUsMGwwLDI1LjY0OTk5bC0yNS42NSwwbDAsLTI1LjY0OTk5eiIvPgogIDxwYXRoIGlkPSJzdmdfMTgiIGZpbGw9IiMzNmNmZDEiIGQ9Im0wLDY0LjE5bDI1LjY1LDBsMCwyNS42NWwtMjUuNjUsMGwwLC0yNS42NXoiLz4KICA8cGF0aCBpZD0ic3ZnXzE5IiBmaWxsPSIjNjI0YWZmIiBkPSJtMTk4LjI4LDg5Ljg0bDI1LjY0OTk5LDBsMCwyNS42NDk5OWwtMjUuNjQ5OTksMGwwLC0yNS42NDk5OXoiLz4KICA8cGF0aCBpZD0ic3ZnXzIwIiBmaWxsPSIjMzZjZmQxIiBkPSJtMTk4LjI4LDY0LjE5bDI1LjY0OTk5LDBsMCwyNS42NWwtMjUuNjQ5OTksMGwwLC0yNS42NXoiLz4KICA8cGF0aCBpZD0ic3ZnXzIxIiBmaWxsPSIjNjI0YWZmIiBkPSJtMTUwLjQ0LDQybDAsMjIuMTlsMjUuNjQ5OTksMGwwLDI1LjY1bDIyLjE5LDBsMCwtNDcuODRsLTQ3Ljg0LDB6Ii8+CiAgPHBhdGggaWQ9InN2Z18yMiIgZmlsbD0iIzM2Y2ZkMSIgZD0ibTczLjQ5LDg5Ljg0bDI1LjY1LDBsMCwyNS42NDk5OWwtMjUuNjUsMGwwLC0yNS42NDk5OXoiLz4KICA8cGF0aCBpZD0ic3ZnXzIzIiBmaWxsPSIjNjI0YWZmIiBkPSJtNDcuODQsNjQuMTlsMjUuNjUsMGwwLC0yMi4xOWwtNDcuODQsMGwwLDQ3Ljg0bDIyLjE5LDBsMCwtMjUuNjV6Ii8+CiAgPHBhdGggaWQ9InN2Z18yNCIgZmlsbD0iIzYyNGFmZiIgZD0ibTQ3Ljg0LDExNS40OWwtMjIuMTksMGwwLDQ3Ljg0bDQ3Ljg0LDBsMCwtMjIuMTlsLTI1LjY1LDBsMCwtMjUuNjV6Ii8+CiA8L2c+Cjwvc3ZnPg==&labelColor=white)](https://www.modelscope.cn/studios/OpenDataLab/MinerU) ### 使用CPU快速体验 @@ -212,10 +232,18 @@ pip install -U magic-pdf[full] --extra-index-url https://wheels.myhloli.com -i h ```json { - // other config - "table-config": { - "model": "TableMaster", // 使用structEqTable请修改为'struct_eqtable' - "is_table_recog_enable": false, // 表格识别功能默认是关闭的,如果需要修改此处的值 + // other config + "layout-config": { + "model": "layoutlmv3" // 使用doclayout_yolo请修改为“doclayout_yolo" + }, + "formula-config": { + "mfd_model": "yolo_v8_mfd", + "mfr_model": "unimernet_small", + "enable": true // 公式识别功能默认是开启的,如果需要关闭请修改此处的值为"false" + }, + "table-config": { + "model": "tablemaster", // 使用structEqTable请修改为"struct_eqtable" + "enable": false, // 表格识别功能默认是关闭的,如果需要开启请修改此处的值为"true" "max_time": 400 } } @@ -265,8 +293,8 @@ Options: -l, --lang TEXT Input the languages in the pdf (if known) to improve OCR accuracy. Optional. You should input "Abbreviation" with language form url: ht - tps://paddlepaddle.github.io/PaddleOCR/en/ppocr - /blog/multi_languages.html#5-support-languages- + tps://paddlepaddle.github.io/PaddleOCR/latest/en + /ppocr/blog/multi_languages.html#5-support-languages- and-abbreviations -d, --debug BOOLEAN Enables detailed debugging information during the execution of the CLI commands. @@ -290,7 +318,7 @@ magic-pdf -p {some_pdf} -o {some_output_dir} -m auto ```text ├── some_pdf.md # markdown 文件 ├── images # 存放图片目录 -├── some_pdf_layout.pdf # layout 绘图 +├── some_pdf_layout.pdf # layout 绘图 (包含layout阅读顺序) ├── some_pdf_middle.json # minerU 中间处理结果 ├── some_pdf_model.json # 模型推理结果 ├── some_pdf_origin.pdf # 原 pdf 文件 @@ -335,29 +363,38 @@ md_content = pipe.pipe_mk_markdown(image_dir, drop_mode="none") - [demo.py 最简单的处理方式](demo/demo.py) - [magic_pdf_parse_main.py 能够更清晰看到处理流程](demo/magic_pdf_parse_main.py) +### 部署衍生项目 + +衍生项目包含项目开发者和社群开发者们基于MinerU的二次开发项目, +例如基于Gradio的应用界面、基于llama的RAG、官网同款web demo、轻量级的多卡负载均衡c/s端等, +这些项目可能会提供更多的功能和更好的用户体验。 +具体部署方式请参考 [衍生项目readme](projects/README_zh-CN.md) + + ### 二次开发 TODO # TODO -- [x] 基于语义的阅读顺序 -- [ ] 正文中列表识别 -- [ ] 正文中代码块识别 -- [ ] 目录识别 -- [x] 表格识别 -- [ ] [化学式识别](docs/chemical_knowledge_introduction/introduction.pdf) -- [ ] 几何图形识别 +- 🗹 基于模型的阅读顺序 +- 🗹 正文中目录、列表识别 +- 🗹 表格识别 +- ☐ 正文中代码块识别 +- ☐ [化学式识别](docs/chemical_knowledge_introduction/introduction.pdf) +- ☐ 几何图形识别 # Known Issues -- 阅读顺序基于规则的分割,在一些情况下会乱序 +- 阅读顺序基于模型对可阅读内容在空间中的分布进行排序,在极端复杂的排版下可能会部分区域乱序 - 不支持竖排文字 -- 列表、代码块、目录在layout模型里还没有支持 +- 目录和列表通过规则进行识别,少部分不常见的列表形式可能无法识别 +- 标题只有一级,目前不支持标题分级 +- 代码块在layout模型里还没有支持 - 漫画书、艺术图册、小学教材、习题尚不能很好解析 -- 在一些公式密集的PDF上强制启用OCR效果会更好 -- 如果您要处理包含大量公式的pdf,强烈建议开启OCR功能。使用pymuPDF提取文字的时候会出现文本行互相重叠的情况导致公式插入位置不准确。 - +- 表格识别在复杂表格上可能会出现行/列识别错误 +- 在小语种PDF上,OCR识别可能会出现字符不准确的情况(如拉丁文的重音符号、阿拉伯文易混淆字符等) +- 部分公式可能会无法在markdown中渲染 # FAQ diff --git a/demo/demo.py b/demo/demo.py index eb3eca1b..a3bd6d03 100644 --- a/demo/demo.py +++ b/demo/demo.py @@ -1,35 +1,22 @@ import os -import json from loguru import logger - from magic_pdf.pipe.UNIPipe import UNIPipe from magic_pdf.rw.DiskReaderWriter import DiskReaderWriter -import magic_pdf.model as model_config -model_config.__use_inside_model__ = True try: current_script_dir = os.path.dirname(os.path.abspath(__file__)) demo_name = "demo1" pdf_path = os.path.join(current_script_dir, f"{demo_name}.pdf") - model_path = os.path.join(current_script_dir, f"{demo_name}.json") pdf_bytes = open(pdf_path, "rb").read() - # model_json = json.loads(open(model_path, "r", encoding="utf-8").read()) - model_json = [] # model_json传空list使用内置模型解析 - jso_useful_key = {"_pdf_type": "", "model_list": model_json} + jso_useful_key = {"_pdf_type": "", "model_list": []} local_image_dir = os.path.join(current_script_dir, 'images') image_dir = str(os.path.basename(local_image_dir)) image_writer = DiskReaderWriter(local_image_dir) pipe = UNIPipe(pdf_bytes, jso_useful_key, image_writer) pipe.pipe_classify() - """如果没有传入有效的模型数据,则使用内置model解析""" - if len(model_json) == 0: - if model_config.__use_inside_model__: - pipe.pipe_analyze() - else: - logger.error("need model list input") - exit(1) + pipe.pipe_analyze() pipe.pipe_parse() md_content = pipe.pipe_mk_markdown(image_dir, drop_mode="none") with open(f"{demo_name}.md", "w", encoding="utf-8") as f: diff --git a/demo/magic_pdf_parse_main.py b/demo/magic_pdf_parse_main.py index 95b84ef0..0e5b9331 100644 --- a/demo/magic_pdf_parse_main.py +++ b/demo/magic_pdf_parse_main.py @@ -4,13 +4,12 @@ import copy from loguru import logger +from magic_pdf.libs.draw_bbox import draw_layout_bbox, draw_span_bbox from magic_pdf.pipe.UNIPipe import UNIPipe from magic_pdf.pipe.OCRPipe import OCRPipe from magic_pdf.pipe.TXTPipe import TXTPipe from magic_pdf.rw.DiskReaderWriter import DiskReaderWriter -import magic_pdf.model as model_config -model_config.__use_inside_model__ = True # todo: 设备类型选择 (?) @@ -47,11 +46,20 @@ def json_md_dump( ) +# 可视化 +def draw_visualization_bbox(pdf_info, pdf_bytes, local_md_dir, pdf_file_name): + # 画布局框,附带排序结果 + draw_layout_bbox(pdf_info, pdf_bytes, local_md_dir, pdf_file_name) + # 画 span 框 + draw_span_bbox(pdf_info, pdf_bytes, local_md_dir, pdf_file_name) + + def pdf_parse_main( pdf_path: str, parse_method: str = 'auto', model_json_path: str = None, is_json_md_dump: bool = True, + is_draw_visualization_bbox: bool = True, output_dir: str = None ): """ @@ -108,11 +116,7 @@ def pdf_parse_main( # 如果没有传入模型数据,则使用内置模型解析 if not model_json: - if model_config.__use_inside_model__: - pipe.pipe_analyze() # 解析 - else: - logger.error("need model list input") - exit(1) + pipe.pipe_analyze() # 解析 # 执行解析 pipe.pipe_parse() @@ -121,10 +125,11 @@ def pdf_parse_main( content_list = pipe.pipe_mk_uni_format(image_path_parent, drop_mode="none") md_content = pipe.pipe_mk_markdown(image_path_parent, drop_mode="none") - if is_json_md_dump: json_md_dump(pipe, md_writer, pdf_name, content_list, md_content) + if is_draw_visualization_bbox: + draw_visualization_bbox(pipe.pdf_mid_data['pdf_info'], pdf_bytes, output_path, pdf_name) except Exception as e: logger.exception(e) @@ -132,5 +137,5 @@ def pdf_parse_main( # 测试 if __name__ == '__main__': - pdf_path = r"C:\Users\XYTK2\Desktop\2024-2016-gb-cd-300.pdf" + pdf_path = r"D:\project\20240617magicpdf\Magic-PDF\demo\demo1.pdf" pdf_parse_main(pdf_path) diff --git a/old_docs/FAQ_en_us.md b/docs/FAQ_en_us.md similarity index 100% rename from old_docs/FAQ_en_us.md rename to docs/FAQ_en_us.md diff --git a/old_docs/FAQ_zh_cn.md b/docs/FAQ_zh_cn.md similarity index 100% rename from old_docs/FAQ_zh_cn.md rename to docs/FAQ_zh_cn.md diff --git a/old_docs/README_Ubuntu_CUDA_Acceleration_en_US.md b/docs/README_Ubuntu_CUDA_Acceleration_en_US.md similarity index 94% rename from old_docs/README_Ubuntu_CUDA_Acceleration_en_US.md rename to docs/README_Ubuntu_CUDA_Acceleration_en_US.md index 23adfea4..010cc729 100644 --- a/old_docs/README_Ubuntu_CUDA_Acceleration_en_US.md +++ b/docs/README_Ubuntu_CUDA_Acceleration_en_US.md @@ -8,6 +8,8 @@ nvidia-smi If you see information similar to the following, it means that the NVIDIA drivers are already installed, and you can skip Step 2. +Notice:`CUDA Version` should be >= 12.1, If the displayed version number is less than 12.1, please upgrade the driver. + ```plaintext +---------------------------------------------------------------------------------------+ | NVIDIA-SMI 537.34 Driver Version: 537.34 CUDA Version: 12.2 | @@ -95,8 +97,6 @@ magic-pdf -p small_ocr.pdf If your graphics card has at least **8GB** of VRAM, follow these steps to test CUDA acceleration: -> ❗ Due to the extremely limited nature of 8GB VRAM for running this application, you need to close all other programs using VRAM to ensure that 8GB of VRAM is available when running this application. - 1. Modify the value of `"device-mode"` in the `magic-pdf.json` configuration file located in your home directory. ```json { diff --git a/old_docs/README_Ubuntu_CUDA_Acceleration_zh_CN.md b/docs/README_Ubuntu_CUDA_Acceleration_zh_CN.md similarity index 95% rename from old_docs/README_Ubuntu_CUDA_Acceleration_zh_CN.md rename to docs/README_Ubuntu_CUDA_Acceleration_zh_CN.md index ebef3255..f72ffec8 100644 --- a/old_docs/README_Ubuntu_CUDA_Acceleration_zh_CN.md +++ b/docs/README_Ubuntu_CUDA_Acceleration_zh_CN.md @@ -8,6 +8,9 @@ nvidia-smi 如果看到类似如下的信息,说明已经安装了nvidia驱动,可以跳过步骤2 +注意:`CUDA Version` 显示的版本号应 >= 12.1,如显示的版本号小于12.1,请升级驱动 + +```plaintext ``` +---------------------------------------------------------------------------------------+ | NVIDIA-SMI 537.34 Driver Version: 537.34 CUDA Version: 12.2 | @@ -95,8 +98,6 @@ magic-pdf -p small_ocr.pdf 如果您的显卡显存大于等于 **8GB** ,可以进行以下流程,测试CUDA解析加速效果 -> ❗️因8GB显存运行本应用非常极限,需要关闭所有其他正在使用显存的程序以确保本应用运行时有足额8GB显存可用。 - **1.修改【用户目录】中配置文件magic-pdf.json中"device-mode"的值** ```json diff --git a/old_docs/README_Windows_CUDA_Acceleration_en_US.md b/docs/README_Windows_CUDA_Acceleration_en_US.md similarity index 93% rename from old_docs/README_Windows_CUDA_Acceleration_en_US.md rename to docs/README_Windows_CUDA_Acceleration_en_US.md index c170cc08..e258011d 100644 --- a/old_docs/README_Windows_CUDA_Acceleration_en_US.md +++ b/docs/README_Windows_CUDA_Acceleration_en_US.md @@ -60,8 +60,6 @@ Download a sample file from the repository and test it. If your graphics card has at least 8GB of VRAM, follow these steps to test CUDA-accelerated parsing performance. -> ❗ Due to the extremely limited nature of 8GB VRAM for running this application, you need to close all other programs using VRAM to ensure that 8GB of VRAM is available when running this application. - 1. **Overwrite the installation of torch and torchvision** supporting CUDA. ``` diff --git a/old_docs/README_Windows_CUDA_Acceleration_zh_CN.md b/docs/README_Windows_CUDA_Acceleration_zh_CN.md similarity index 94% rename from old_docs/README_Windows_CUDA_Acceleration_zh_CN.md rename to docs/README_Windows_CUDA_Acceleration_zh_CN.md index 8d2457a6..a88a50db 100644 --- a/old_docs/README_Windows_CUDA_Acceleration_zh_CN.md +++ b/docs/README_Windows_CUDA_Acceleration_zh_CN.md @@ -61,8 +61,6 @@ pip install -U magic-pdf[full] --extra-index-url https://wheels.myhloli.com -i h 如果您的显卡显存大于等于 **8GB** ,可以进行以下流程,测试CUDA解析加速效果 -> ❗️因8GB显存运行本应用非常极限,需要关闭所有其他正在使用显存的程序以确保本应用运行时有足额8GB显存可用。 - **1.覆盖安装支持cuda的torch和torchvision** ```bash diff --git a/old_docs/chemical_knowledge_introduction/introduction.pdf b/docs/chemical_knowledge_introduction/introduction.pdf similarity index 100% rename from old_docs/chemical_knowledge_introduction/introduction.pdf rename to docs/chemical_knowledge_introduction/introduction.pdf diff --git a/old_docs/chemical_knowledge_introduction/introduction.xmind b/docs/chemical_knowledge_introduction/introduction.xmind similarity index 100% rename from old_docs/chemical_knowledge_introduction/introduction.xmind rename to docs/chemical_knowledge_introduction/introduction.xmind diff --git a/old_docs/download_models.py b/docs/download_models.py similarity index 59% rename from old_docs/download_models.py rename to docs/download_models.py index 7f116a0c..ed1ee5c3 100644 --- a/old_docs/download_models.py +++ b/docs/download_models.py @@ -5,16 +5,21 @@ import requests from modelscope import snapshot_download +def download_json(url): + # 下载JSON文件 + response = requests.get(url) + response.raise_for_status() # 检查请求是否成功 + return response.json() + + def download_and_modify_json(url, local_filename, modifications): if os.path.exists(local_filename): data = json.load(open(local_filename)) + config_version = data.get('config_version', '0.0.0') + if config_version < '1.0.0': + data = download_json(url) else: - # 下载JSON文件 - response = requests.get(url) - response.raise_for_status() # 检查请求是否成功 - - # 解析JSON内容 - data = response.json() + data = download_json(url) # 修改内容 for key, value in modifications.items(): @@ -26,13 +31,21 @@ def download_and_modify_json(url, local_filename, modifications): if __name__ == '__main__': - model_dir = snapshot_download('opendatalab/PDF-Extract-Kit') + mineru_patterns = [ + "models/Layout/LayoutLMv3/*", + "models/Layout/YOLO/*", + "models/MFD/YOLO/*", + "models/MFR/unimernet_small/*", + "models/TabRec/TableMaster/*", + "models/TabRec/StructEqTable/*", + ] + model_dir = snapshot_download('opendatalab/PDF-Extract-Kit-1.0', allow_patterns=mineru_patterns) layoutreader_model_dir = snapshot_download('ppaanngggg/layoutreader') model_dir = model_dir + '/models' print(f'model_dir is: {model_dir}') print(f'layoutreader_model_dir is: {layoutreader_model_dir}') - json_url = 'https://gitee.com/myhloli/MinerU/raw/master/magic-pdf.template.json' + json_url = 'https://gitee.com/myhloli/MinerU/raw/dev/magic-pdf.template.json' config_file_name = 'magic-pdf.json' home_dir = os.path.expanduser('~') config_file = os.path.join(home_dir, config_file_name) diff --git a/old_docs/download_models_hf.py b/docs/download_models_hf.py similarity index 55% rename from old_docs/download_models_hf.py rename to docs/download_models_hf.py index 915f1a24..5e6b8dce 100644 --- a/old_docs/download_models_hf.py +++ b/docs/download_models_hf.py @@ -5,16 +5,21 @@ import requests from huggingface_hub import snapshot_download +def download_json(url): + # 下载JSON文件 + response = requests.get(url) + response.raise_for_status() # 检查请求是否成功 + return response.json() + + def download_and_modify_json(url, local_filename, modifications): if os.path.exists(local_filename): data = json.load(open(local_filename)) + config_version = data.get('config_version', '0.0.0') + if config_version < '1.0.0': + data = download_json(url) else: - # 下载JSON文件 - response = requests.get(url) - response.raise_for_status() # 检查请求是否成功 - - # 解析JSON内容 - data = response.json() + data = download_json(url) # 修改内容 for key, value in modifications.items(): @@ -26,13 +31,28 @@ def download_and_modify_json(url, local_filename, modifications): if __name__ == '__main__': - model_dir = snapshot_download('opendatalab/PDF-Extract-Kit') - layoutreader_model_dir = snapshot_download('hantian/layoutreader') + + mineru_patterns = [ + "models/Layout/LayoutLMv3/*", + "models/Layout/YOLO/*", + "models/MFD/YOLO/*", + "models/MFR/unimernet_small/*", + "models/TabRec/TableMaster/*", + "models/TabRec/StructEqTable/*", + ] + model_dir = snapshot_download('opendatalab/PDF-Extract-Kit-1.0', allow_patterns=mineru_patterns) + + layoutreader_pattern = [ + "*.json", + "*.safetensors", + ] + layoutreader_model_dir = snapshot_download('hantian/layoutreader', allow_patterns=layoutreader_pattern) + model_dir = model_dir + '/models' print(f'model_dir is: {model_dir}') print(f'layoutreader_model_dir is: {layoutreader_model_dir}') - json_url = 'https://github.com/opendatalab/MinerU/raw/master/magic-pdf.template.json' + json_url = 'https://github.com/opendatalab/MinerU/raw/dev/magic-pdf.template.json' config_file_name = 'magic-pdf.json' home_dir = os.path.expanduser('~') config_file = os.path.join(home_dir, config_file_name) diff --git a/old_docs/how_to_download_models_en.md b/docs/how_to_download_models_en.md similarity index 73% rename from old_docs/how_to_download_models_en.md rename to docs/how_to_download_models_en.md index 359afdea..34dd54e4 100644 --- a/old_docs/how_to_download_models_en.md +++ b/docs/how_to_download_models_en.md @@ -22,7 +22,9 @@ The configuration file can be found in the user directory, with the filename `ma > Due to feedback from some users that downloading model files using git lfs was incomplete or resulted in corrupted model files, this method is no longer recommended. -If you previously downloaded model files via git lfs, you can navigate to the previous download directory and use the `git pull` command to update the model. +When magic-pdf <= 0.8.1, if you have previously downloaded the model files via git lfs, you can navigate to the previous download directory and update the models using the `git pull` command. + +> For versions 0.9.x and later, due to the repository change and the addition of the layout sorting model in PDF-Extract-Kit 1.0, the models cannot be updated using the `git pull` command. Instead, a Python script must be used for one-click updates. ## 2. Models downloaded via Hugging Face or Model Scope diff --git a/old_docs/how_to_download_models_zh_cn.md b/docs/how_to_download_models_zh_cn.md similarity index 80% rename from old_docs/how_to_download_models_zh_cn.md rename to docs/how_to_download_models_zh_cn.md index 6b0db501..9b395e34 100644 --- a/old_docs/how_to_download_models_zh_cn.md +++ b/docs/how_to_download_models_zh_cn.md @@ -34,14 +34,10 @@ python脚本会自动下载模型文件并配置好配置文件中的模型目 > 由于部分用户反馈通过git lfs下载模型文件遇到下载不全和模型文件损坏情况,现已不推荐使用该方式下载。 -如此前通过 git lfs 下载过模型文件,可以进入到之前的下载目录中,通过`git pull`命令更新模型。 +当magic-pdf <= 0.8.1时,如此前通过 git lfs 下载过模型文件,可以进入到之前的下载目录中,通过`git pull`命令更新模型。 + +> 0.9.x及以后版本由于PDF-Extract-Kit 1.0更换仓库和新增layout排序模型,不能通过`git pull`命令更新,需要使用python脚本一键更新。 -> 0.9.x及以后版本由于新增layout排序模型,且该模型和此前的模型不在同一仓库,不能通过`git pull`命令更新,需要单独下载。 -> -> ``` -> from modelscope import snapshot_download -> snapshot_download('ppaanngggg/layoutreader') -> ``` ## 2. 通过 Hugging Face 或 Model Scope 下载过模型 diff --git a/old_docs/images/MinerU-logo-hq.png b/docs/images/MinerU-logo-hq.png similarity index 100% rename from old_docs/images/MinerU-logo-hq.png rename to docs/images/MinerU-logo-hq.png diff --git a/old_docs/images/MinerU-logo.png b/docs/images/MinerU-logo.png similarity index 100% rename from old_docs/images/MinerU-logo.png rename to docs/images/MinerU-logo.png diff --git a/old_docs/images/datalab_logo.png b/docs/images/datalab_logo.png similarity index 100% rename from old_docs/images/datalab_logo.png rename to docs/images/datalab_logo.png diff --git a/old_docs/images/flowchart_en.png b/docs/images/flowchart_en.png similarity index 100% rename from old_docs/images/flowchart_en.png rename to docs/images/flowchart_en.png diff --git a/old_docs/images/flowchart_zh_cn.png b/docs/images/flowchart_zh_cn.png similarity index 100% rename from old_docs/images/flowchart_zh_cn.png rename to docs/images/flowchart_zh_cn.png diff --git a/old_docs/images/layout_example.png b/docs/images/layout_example.png similarity index 100% rename from old_docs/images/layout_example.png rename to docs/images/layout_example.png diff --git a/old_docs/images/poly.png b/docs/images/poly.png similarity index 100% rename from old_docs/images/poly.png rename to docs/images/poly.png diff --git a/old_docs/images/project_panorama_en.png b/docs/images/project_panorama_en.png similarity index 100% rename from old_docs/images/project_panorama_en.png rename to docs/images/project_panorama_en.png diff --git a/old_docs/images/project_panorama_zh_cn.png b/docs/images/project_panorama_zh_cn.png similarity index 100% rename from old_docs/images/project_panorama_zh_cn.png rename to docs/images/project_panorama_zh_cn.png diff --git a/old_docs/images/spans_example.png b/docs/images/spans_example.png similarity index 100% rename from old_docs/images/spans_example.png rename to docs/images/spans_example.png diff --git a/old_docs/images/web_demo_1.png b/docs/images/web_demo_1.png similarity index 100% rename from old_docs/images/web_demo_1.png rename to docs/images/web_demo_1.png diff --git a/old_docs/output_file_en_us.md b/docs/output_file_en_us.md similarity index 100% rename from old_docs/output_file_en_us.md rename to docs/output_file_en_us.md diff --git a/old_docs/output_file_zh_cn.md b/docs/output_file_zh_cn.md similarity index 100% rename from old_docs/output_file_zh_cn.md rename to docs/output_file_zh_cn.md diff --git a/magic-pdf.template.json b/magic-pdf.template.json index fcb99955..114dfce3 100644 --- a/magic-pdf.template.json +++ b/magic-pdf.template.json @@ -6,9 +6,18 @@ "models-dir":"/tmp/models", "layoutreader-model-dir":"/tmp/layoutreader", "device-mode":"cpu", + "layout-config": { + "model": "layoutlmv3" + }, + "formula-config": { + "mfd_model": "yolo_v8_mfd", + "mfr_model": "unimernet_small", + "enable": true + }, "table-config": { - "model": "TableMaster", - "is_table_recog_enable": false, + "model": "tablemaster", + "enable": false, "max_time": 400 - } + }, + "config_version": "1.0.0" } \ No newline at end of file diff --git a/magic_pdf/config/__init__.py b/magic_pdf/config/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/magic_pdf/config/enums.py b/magic_pdf/config/enums.py new file mode 100644 index 00000000..6f3e91a3 --- /dev/null +++ b/magic_pdf/config/enums.py @@ -0,0 +1,7 @@ + +import enum + + +class SupportedPdfParseMethod(enum.Enum): + OCR = 'ocr' + TXT = 'txt' diff --git a/magic_pdf/config/exceptions.py b/magic_pdf/config/exceptions.py new file mode 100644 index 00000000..38f326b5 --- /dev/null +++ b/magic_pdf/config/exceptions.py @@ -0,0 +1,32 @@ + +class FileNotExisted(Exception): + + def __init__(self, path): + self.path = path + + def __str__(self): + return f'File {self.path} does not exist.' + + +class InvalidConfig(Exception): + def __init__(self, msg): + self.msg = msg + + def __str__(self): + return f'Invalid config: {self.msg}' + + +class InvalidParams(Exception): + def __init__(self, msg): + self.msg = msg + + def __str__(self): + return f'Invalid params: {self.msg}' + + +class EmptyData(Exception): + def __init__(self, msg): + self.msg = msg + + def __str__(self): + return f'Empty data: {self.msg}' diff --git a/magic_pdf/data/__init__.py b/magic_pdf/data/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/magic_pdf/data/data_reader_writer/__init__.py b/magic_pdf/data/data_reader_writer/__init__.py new file mode 100644 index 00000000..f8f82347 --- /dev/null +++ b/magic_pdf/data/data_reader_writer/__init__.py @@ -0,0 +1,12 @@ +from magic_pdf.data.data_reader_writer.filebase import \ + FileBasedDataReader # noqa: F401 +from magic_pdf.data.data_reader_writer.filebase import \ + FileBasedDataWriter # noqa: F401 +from magic_pdf.data.data_reader_writer.multi_bucket_s3 import \ + MultiBucketS3DataReader # noqa: F401 +from magic_pdf.data.data_reader_writer.multi_bucket_s3 import \ + MultiBucketS3DataWriter # noqa: F401 +from magic_pdf.data.data_reader_writer.s3 import S3DataReader # noqa: F401 +from magic_pdf.data.data_reader_writer.s3 import S3DataWriter # noqa: F401 +from magic_pdf.data.data_reader_writer.base import DataReader # noqa: F401 +from magic_pdf.data.data_reader_writer.base import DataWriter # noqa: F401 \ No newline at end of file diff --git a/magic_pdf/data/data_reader_writer/base.py b/magic_pdf/data/data_reader_writer/base.py new file mode 100644 index 00000000..7c9a8e8e --- /dev/null +++ b/magic_pdf/data/data_reader_writer/base.py @@ -0,0 +1,51 @@ + +from abc import ABC, abstractmethod + + +class DataReader(ABC): + + def read(self, path: str) -> bytes: + """Read the file. + + Args: + path (str): file path to read + + Returns: + bytes: the content of the file + """ + return self.read_at(path) + + @abstractmethod + def read_at(self, path: str, offset: int = 0, limit: int = -1) -> bytes: + """Read the file at offset and limit. + + Args: + path (str): the file path + offset (int, optional): the number of bytes skipped. Defaults to 0. + limit (int, optional): the length of bytes want to read. Defaults to -1. + + Returns: + bytes: the content of the file + """ + pass + + +class DataWriter(ABC): + @abstractmethod + def write(self, path: str, data: bytes) -> None: + """Write the data to the file. + + Args: + path (str): the target file where to write + data (bytes): the data want to write + """ + pass + + def write_string(self, path: str, data: str) -> None: + """Write the data to file, the data will be encoded to bytes. + + Args: + path (str): the target file where to write + data (str): the data want to write + """ + self.write(path, data.encode()) diff --git a/magic_pdf/data/data_reader_writer/filebase.py b/magic_pdf/data/data_reader_writer/filebase.py new file mode 100644 index 00000000..40e9e4ff --- /dev/null +++ b/magic_pdf/data/data_reader_writer/filebase.py @@ -0,0 +1,59 @@ +import os + +from magic_pdf.data.data_reader_writer.base import DataReader, DataWriter + + +class FileBasedDataReader(DataReader): + def __init__(self, parent_dir: str = ''): + """Initialized with parent_dir. + + Args: + parent_dir (str, optional): the parent directory that may be used within methods. Defaults to ''. + """ + self._parent_dir = parent_dir + + def read_at(self, path: str, offset: int = 0, limit: int = -1) -> bytes: + """Read at offset and limit. + + Args: + path (str): the path of file, if the path is relative path, it will be joined with parent_dir. + offset (int, optional): the number of bytes skipped. Defaults to 0. + limit (int, optional): the length of bytes want to read. Defaults to -1. + + Returns: + bytes: the content of file + """ + fn_path = path + if not os.path.isabs(fn_path) and len(self._parent_dir) > 0: + fn_path = os.path.join(self._parent_dir, path) + + with open(fn_path, 'rb') as f: + f.seek(offset) + if limit == -1: + return f.read() + else: + return f.read(limit) + + +class FileBasedDataWriter(DataWriter): + def __init__(self, parent_dir: str = '') -> None: + """Initialized with parent_dir. + + Args: + parent_dir (str, optional): the parent directory that may be used within methods. Defaults to ''. + """ + self._parent_dir = parent_dir + + def write(self, path: str, data: bytes) -> None: + """Write file with data. + + Args: + path (str): the path of file, if the path is relative path, it will be joined with parent_dir. + data (bytes): the data want to write + """ + fn_path = path + if not os.path.isabs(fn_path) and len(self._parent_dir) > 0: + fn_path = os.path.join(self._parent_dir, path) + + with open(fn_path, 'wb') as f: + f.write(data) diff --git a/magic_pdf/data/data_reader_writer/multi_bucket_s3.py b/magic_pdf/data/data_reader_writer/multi_bucket_s3.py new file mode 100644 index 00000000..4f6347b3 --- /dev/null +++ b/magic_pdf/data/data_reader_writer/multi_bucket_s3.py @@ -0,0 +1,137 @@ +from magic_pdf.config.exceptions import InvalidConfig, InvalidParams +from magic_pdf.data.data_reader_writer.base import DataReader, DataWriter +from magic_pdf.data.io.s3 import S3Reader, S3Writer +from magic_pdf.data.schemas import S3Config +from magic_pdf.libs.path_utils import (parse_s3_range_params, parse_s3path, + remove_non_official_s3_args) + + +class MultiS3Mixin: + def __init__(self, default_bucket: str, s3_configs: list[S3Config]): + """Initialized with multiple s3 configs. + + Args: + default_bucket (str): the default bucket name of the relative path + s3_configs (list[S3Config]): list of s3 configs, the bucket_name must be unique in the list. + + Raises: + InvalidConfig: default bucket config not in s3_configs + InvalidConfig: bucket name not unique in s3_configs + InvalidConfig: default bucket must be provided + """ + if len(default_bucket) == 0: + raise InvalidConfig('default_bucket must be provided') + + found_default_bucket_config = False + for conf in s3_configs: + if conf.bucket_name == default_bucket: + found_default_bucket_config = True + break + + if not found_default_bucket_config: + raise InvalidConfig( + f'default_bucket: {default_bucket} config must be provided in s3_configs: {s3_configs}' + ) + + uniq_bucket = set([conf.bucket_name for conf in s3_configs]) + if len(uniq_bucket) != len(s3_configs): + raise InvalidConfig( + f'the bucket_name in s3_configs: {s3_configs} must be unique' + ) + + self.default_bucket = default_bucket + self.s3_configs = s3_configs + self._s3_clients_h: dict = {} + + +class MultiBucketS3DataReader(DataReader, MultiS3Mixin): + def read(self, path: str) -> bytes: + """Read the path from s3, select diffect bucket client for each request + based on the path, also support range read. + + Args: + path (str): the s3 path of file, the path must be in the format of s3://bucket_name/path?offset,limit + for example: s3://bucket_name/path?0,100 + + Returns: + bytes: the content of s3 file + """ + may_range_params = parse_s3_range_params(path) + if may_range_params is None or 2 != len(may_range_params): + byte_start, byte_len = 0, -1 + else: + byte_start, byte_len = int(may_range_params[0]), int(may_range_params[1]) + path = remove_non_official_s3_args(path) + return self.read_at(path, byte_start, byte_len) + + def __get_s3_client(self, bucket_name: str): + if bucket_name not in set([conf.bucket_name for conf in self.s3_configs]): + raise InvalidParams( + f'bucket name: {bucket_name} not found in s3_configs: {self.s3_configs}' + ) + if bucket_name not in self._s3_clients_h: + conf = next( + filter(lambda conf: conf.bucket_name == bucket_name, self.s3_configs) + ) + self._s3_clients_h[bucket_name] = S3Reader( + bucket_name, + conf.access_key, + conf.secret_key, + conf.endpoint_url, + conf.addressing_style, + ) + return self._s3_clients_h[bucket_name] + + def read_at(self, path: str, offset: int = 0, limit: int = -1) -> bytes: + """Read the file with offset and limit, select diffect bucket client + for each request based on the path. + + Args: + path (str): the file path + offset (int, optional): the number of bytes skipped. Defaults to 0. + limit (int, optional): the number of bytes want to read. Defaults to -1 which means infinite. + + Returns: + bytes: the file content + """ + if path.startswith('s3://'): + bucket_name, path = parse_s3path(path) + s3_reader = self.__get_s3_client(bucket_name) + else: + s3_reader = self.__get_s3_client(self.default_bucket) + return s3_reader.read_at(path, offset, limit) + + +class MultiBucketS3DataWriter(DataWriter, MultiS3Mixin): + def __get_s3_client(self, bucket_name: str): + if bucket_name not in set([conf.bucket_name for conf in self.s3_configs]): + raise InvalidParams( + f'bucket name: {bucket_name} not found in s3_configs: {self.s3_configs}' + ) + if bucket_name not in self._s3_clients_h: + conf = next( + filter(lambda conf: conf.bucket_name == bucket_name, self.s3_configs) + ) + self._s3_clients_h[bucket_name] = S3Writer( + bucket_name, + conf.access_key, + conf.secret_key, + conf.endpoint_url, + conf.addressing_style, + ) + return self._s3_clients_h[bucket_name] + + def write(self, path: str, data: bytes) -> None: + """Write file with data, also select diffect bucket client for each + request based on the path. + + Args: + path (str): the path of file, if the path is relative path, it will be joined with parent_dir. + data (bytes): the data want to write + """ + if path.startswith('s3://'): + bucket_name, path = parse_s3path(path) + s3_writer = self.__get_s3_client(bucket_name) + else: + s3_writer = self.__get_s3_client(self.default_bucket) + return s3_writer.write(path, data) diff --git a/magic_pdf/data/data_reader_writer/s3.py b/magic_pdf/data/data_reader_writer/s3.py new file mode 100644 index 00000000..b6f27a27 --- /dev/null +++ b/magic_pdf/data/data_reader_writer/s3.py @@ -0,0 +1,69 @@ +from magic_pdf.data.data_reader_writer.multi_bucket_s3 import ( + MultiBucketS3DataReader, MultiBucketS3DataWriter) +from magic_pdf.data.schemas import S3Config + + +class S3DataReader(MultiBucketS3DataReader): + def __init__( + self, + bucket: str, + ak: str, + sk: str, + endpoint_url: str, + addressing_style: str = 'auto', + ): + """s3 reader client. + + Args: + bucket (str): bucket name + ak (str): access key + sk (str): secret key + endpoint_url (str): endpoint url of s3 + addressing_style (str, optional): Defaults to 'auto'. Other valid options here are 'path' and 'virtual' + refer to https://boto3.amazonaws.com/v1/documentation/api/1.9.42/guide/s3.html + """ + super().__init__( + bucket, + [ + S3Config( + bucket_name=bucket, + access_key=ak, + secret_key=sk, + endpoint_url=endpoint_url, + addressing_style=addressing_style, + ) + ], + ) + + +class S3DataWriter(MultiBucketS3DataWriter): + def __init__( + self, + bucket: str, + ak: str, + sk: str, + endpoint_url: str, + addressing_style: str = 'auto', + ): + """s3 writer client. + + Args: + bucket (str): bucket name + ak (str): access key + sk (str): secret key + endpoint_url (str): endpoint url of s3 + addressing_style (str, optional): Defaults to 'auto'. Other valid options here are 'path' and 'virtual' + refer to https://boto3.amazonaws.com/v1/documentation/api/1.9.42/guide/s3.html + """ + super().__init__( + bucket, + [ + S3Config( + bucket_name=bucket, + access_key=ak, + secret_key=sk, + endpoint_url=endpoint_url, + addressing_style=addressing_style, + ) + ], + ) diff --git a/magic_pdf/data/dataset.py b/magic_pdf/data/dataset.py new file mode 100644 index 00000000..0eee3c68 --- /dev/null +++ b/magic_pdf/data/dataset.py @@ -0,0 +1,194 @@ +from abc import ABC, abstractmethod +from typing import Iterator + +import fitz + +from magic_pdf.config.enums import SupportedPdfParseMethod +from magic_pdf.data.schemas import PageInfo +from magic_pdf.data.utils import fitz_doc_to_image + + +class PageableData(ABC): + @abstractmethod + def get_image(self) -> dict: + """Transform data to image.""" + pass + + @abstractmethod + def get_doc(self) -> fitz.Page: + """Get the pymudoc page.""" + pass + + @abstractmethod + def get_page_info(self) -> PageInfo: + """Get the page info of the page. + + Returns: + PageInfo: the page info of this page + """ + pass + + +class Dataset(ABC): + @abstractmethod + def __len__(self) -> int: + """The length of the dataset.""" + pass + + @abstractmethod + def __iter__(self) -> Iterator[PageableData]: + """Yield the page data.""" + pass + + @abstractmethod + def supported_methods(self) -> list[SupportedPdfParseMethod]: + """The methods that this dataset support. + + Returns: + list[SupportedPdfParseMethod]: The supported methods, Valid methods are: OCR, TXT + """ + pass + + @abstractmethod + def data_bits(self) -> bytes: + """The bits used to create this dataset.""" + pass + + @abstractmethod + def get_page(self, page_id: int) -> PageableData: + """Get the page indexed by page_id. + + Args: + page_id (int): the index of the page + + Returns: + PageableData: the page doc object + """ + pass + + +class PymuDocDataset(Dataset): + def __init__(self, bits: bytes): + """Initialize the dataset, which wraps the pymudoc documents. + + Args: + bits (bytes): the bytes of the pdf + """ + self._records = [Doc(v) for v in fitz.open('pdf', bits)] + self._data_bits = bits + self._raw_data = bits + + def __len__(self) -> int: + """The page number of the pdf.""" + return len(self._records) + + def __iter__(self) -> Iterator[PageableData]: + """Yield the page doc object.""" + return iter(self._records) + + def supported_methods(self) -> list[SupportedPdfParseMethod]: + """The method supported by this dataset. + + Returns: + list[SupportedPdfParseMethod]: the supported methods + """ + return [SupportedPdfParseMethod.OCR, SupportedPdfParseMethod.TXT] + + def data_bits(self) -> bytes: + """The pdf bits used to create this dataset.""" + return self._data_bits + + def get_page(self, page_id: int) -> PageableData: + """The page doc object. + + Args: + page_id (int): the page doc index + + Returns: + PageableData: the page doc object + """ + return self._records[page_id] + + +class ImageDataset(Dataset): + def __init__(self, bits: bytes): + """Initialize the dataset, which wraps the pymudoc documents. + + Args: + bits (bytes): the bytes of the photo which will be converted to pdf first. then converted to pymudoc. + """ + pdf_bytes = fitz.open(stream=bits).convert_to_pdf() + self._records = [Doc(v) for v in fitz.open('pdf', pdf_bytes)] + self._raw_data = bits + self._data_bits = pdf_bytes + + def __len__(self) -> int: + """The length of the dataset.""" + return len(self._records) + + def __iter__(self) -> Iterator[PageableData]: + """Yield the page object.""" + return iter(self._records) + + def supported_methods(self): + """The method supported by this dataset. + + Returns: + list[SupportedPdfParseMethod]: the supported methods + """ + return [SupportedPdfParseMethod.OCR] + + def data_bits(self) -> bytes: + """The pdf bits used to create this dataset.""" + return self._data_bits + + def get_page(self, page_id: int) -> PageableData: + """The page doc object. + + Args: + page_id (int): the page doc index + + Returns: + PageableData: the page doc object + """ + return self._records[page_id] + + +class Doc(PageableData): + """Initialized with pymudoc object.""" + def __init__(self, doc: fitz.Page): + self._doc = doc + + def get_image(self): + """Return the imge info. + + Returns: + dict: { + img: np.ndarray, + width: int, + height: int + } + """ + return fitz_doc_to_image(self._doc) + + def get_doc(self) -> fitz.Page: + """Get the pymudoc object. + + Returns: + fitz.Page: the pymudoc object + """ + return self._doc + + def get_page_info(self) -> PageInfo: + """Get the page info of the page. + + Returns: + PageInfo: the page info of this page + """ + page_w = self._doc.rect.width + page_h = self._doc.rect.height + return PageInfo(w=page_w, h=page_h) + + def __getattr__(self, name): + if hasattr(self._doc, name): + return getattr(self._doc, name) diff --git a/magic_pdf/data/io/__init__.py b/magic_pdf/data/io/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/magic_pdf/data/io/base.py b/magic_pdf/data/io/base.py new file mode 100644 index 00000000..c02c2ed4 --- /dev/null +++ b/magic_pdf/data/io/base.py @@ -0,0 +1,42 @@ +from abc import ABC, abstractmethod + + +class IOReader(ABC): + @abstractmethod + def read(self, path: str) -> bytes: + """Read the file. + + Args: + path (str): file path to read + + Returns: + bytes: the content of the file + """ + pass + + @abstractmethod + def read_at(self, path: str, offset: int = 0, limit: int = -1) -> bytes: + """Read at offset and limit. + + Args: + path (str): the path of file, if the path is relative path, it will be joined with parent_dir. + offset (int, optional): the number of bytes skipped. Defaults to 0. + limit (int, optional): the length of bytes want to read. Defaults to -1. + + Returns: + bytes: the content of file + """ + pass + + +class IOWriter: + + @abstractmethod + def write(self, path: str, data: bytes) -> None: + """Write file with data. + + Args: + path (str): the path of file, if the path is relative path, it will be joined with parent_dir. + data (bytes): the data want to write + """ + pass diff --git a/magic_pdf/data/io/http.py b/magic_pdf/data/io/http.py new file mode 100644 index 00000000..3b08271f --- /dev/null +++ b/magic_pdf/data/io/http.py @@ -0,0 +1,37 @@ + +import io + +import requests + +from magic_pdf.data.io.base import IOReader, IOWriter + + +class HttpReader(IOReader): + + def read(self, url: str) -> bytes: + """Read the file. + + Args: + path (str): file path to read + + Returns: + bytes: the content of the file + """ + return requests.get(url).content + + def read_at(self, path: str, offset: int = 0, limit: int = -1) -> bytes: + """Not Implemented.""" + raise NotImplementedError + + +class HttpWriter(IOWriter): + def write(self, url: str, data: bytes) -> None: + """Write file with data. + + Args: + path (str): the path of file, if the path is relative path, it will be joined with parent_dir. + data (bytes): the data want to write + """ + files = {'file': io.BytesIO(data)} + response = requests.post(url, files=files) + assert 300 > response.status_code and response.status_code > 199 diff --git a/magic_pdf/data/io/s3.py b/magic_pdf/data/io/s3.py new file mode 100644 index 00000000..4222c73f --- /dev/null +++ b/magic_pdf/data/io/s3.py @@ -0,0 +1,114 @@ +import boto3 +from botocore.config import Config + +from magic_pdf.data.io.base import IOReader, IOWriter + + +class S3Reader(IOReader): + def __init__( + self, + bucket: str, + ak: str, + sk: str, + endpoint_url: str, + addressing_style: str = 'auto', + ): + """s3 reader client. + + Args: + bucket (str): bucket name + ak (str): access key + sk (str): secret key + endpoint_url (str): endpoint url of s3 + addressing_style (str, optional): Defaults to 'auto'. Other valid options here are 'path' and 'virtual' + refer to https://boto3.amazonaws.com/v1/documentation/api/1.9.42/guide/s3.html + """ + self._bucket = bucket + self._ak = ak + self._sk = sk + self._s3_client = boto3.client( + service_name='s3', + aws_access_key_id=ak, + aws_secret_access_key=sk, + endpoint_url=endpoint_url, + config=Config( + s3={'addressing_style': addressing_style}, + retries={'max_attempts': 5, 'mode': 'standard'}, + ), + ) + + def read(self, key: str) -> bytes: + """Read the file. + + Args: + path (str): file path to read + + Returns: + bytes: the content of the file + """ + return self.read_at(key) + + def read_at(self, key: str, offset: int = 0, limit: int = -1) -> bytes: + """Read at offset and limit. + + Args: + path (str): the path of file, if the path is relative path, it will be joined with parent_dir. + offset (int, optional): the number of bytes skipped. Defaults to 0. + limit (int, optional): the length of bytes want to read. Defaults to -1. + + Returns: + bytes: the content of file + """ + if limit > -1: + range_header = f'bytes={offset}-{offset+limit-1}' + res = self._s3_client.get_object( + Bucket=self._bucket, Key=key, Range=range_header + ) + else: + res = self._s3_client.get_object( + Bucket=self._bucket, Key=key, Range=f'bytes={offset}-' + ) + return res['Body'].read() + + +class S3Writer(IOWriter): + def __init__( + self, + bucket: str, + ak: str, + sk: str, + endpoint_url: str, + addressing_style: str = 'auto', + ): + """s3 reader client. + + Args: + bucket (str): bucket name + ak (str): access key + sk (str): secret key + endpoint_url (str): endpoint url of s3 + addressing_style (str, optional): Defaults to 'auto'. Other valid options here are 'path' and 'virtual' + refer to https://boto3.amazonaws.com/v1/documentation/api/1.9.42/guide/s3.html + """ + self._bucket = bucket + self._ak = ak + self._sk = sk + self._s3_client = boto3.client( + service_name='s3', + aws_access_key_id=ak, + aws_secret_access_key=sk, + endpoint_url=endpoint_url, + config=Config( + s3={'addressing_style': addressing_style}, + retries={'max_attempts': 5, 'mode': 'standard'}, + ), + ) + + def write(self, key: str, data: bytes): + """Write file with data. + + Args: + path (str): the path of file, if the path is relative path, it will be joined with parent_dir. + data (bytes): the data want to write + """ + self._s3_client.put_object(Bucket=self._bucket, Key=key, Body=data) diff --git a/magic_pdf/data/read_api.py b/magic_pdf/data/read_api.py new file mode 100644 index 00000000..61e0fbf7 --- /dev/null +++ b/magic_pdf/data/read_api.py @@ -0,0 +1,95 @@ +import json +import os +from pathlib import Path + +from magic_pdf.config.exceptions import EmptyData, InvalidParams +from magic_pdf.data.data_reader_writer import (FileBasedDataReader, + MultiBucketS3DataReader) +from magic_pdf.data.dataset import ImageDataset, PymuDocDataset + + +def read_jsonl( + s3_path_or_local: str, s3_client: MultiBucketS3DataReader | None = None +) -> list[PymuDocDataset]: + """Read the jsonl file and return the list of PymuDocDataset. + + Args: + s3_path_or_local (str): local file or s3 path + s3_client (MultiBucketS3DataReader | None, optional): s3 client that support multiple bucket. Defaults to None. + + Raises: + InvalidParams: if s3_path_or_local is s3 path but s3_client is not provided. + EmptyData: if no pdf file location is provided in some line of jsonl file. + InvalidParams: if the file location is s3 path but s3_client is not provided + + Returns: + list[PymuDocDataset]: each line in the jsonl file will be converted to a PymuDocDataset + """ + bits_arr = [] + if s3_path_or_local.startswith('s3://'): + if s3_client is None: + raise InvalidParams('s3_client is required when s3_path is provided') + jsonl_bits = s3_client.read(s3_path_or_local) + else: + jsonl_bits = FileBasedDataReader('').read(s3_path_or_local) + jsonl_d = [ + json.loads(line) for line in jsonl_bits.decode().split('\n') if line.strip() + ] + for d in jsonl_d[:5]: + pdf_path = d.get('file_location', '') or d.get('path', '') + if len(pdf_path) == 0: + raise EmptyData('pdf file location is empty') + if pdf_path.startswith('s3://'): + if s3_client is None: + raise InvalidParams('s3_client is required when s3_path is provided') + bits_arr.append(s3_client.read(pdf_path)) + else: + bits_arr.append(FileBasedDataReader('').read(pdf_path)) + return [PymuDocDataset(bits) for bits in bits_arr] + + +def read_local_pdfs(path: str) -> list[PymuDocDataset]: + """Read pdf from path or directory. + + Args: + path (str): pdf file path or directory that contains pdf files + + Returns: + list[PymuDocDataset]: each pdf file will converted to a PymuDocDataset + """ + if os.path.isdir(path): + reader = FileBasedDataReader(path) + return [ + PymuDocDataset(reader.read(doc_path.name)) + for doc_path in Path(path).glob('*.pdf') + ] + else: + reader = FileBasedDataReader() + bits = reader.read(path) + return [PymuDocDataset(bits)] + + +def read_local_images(path: str, suffixes: list[str]) -> list[ImageDataset]: + """Read images from path or directory. + + Args: + path (str): image file path or directory that contains image files + suffixes (list[str]): the suffixes of the image files used to filter the files. Example: ['jpg', 'png'] + + Returns: + list[ImageDataset]: each image file will converted to a ImageDataset + """ + if os.path.isdir(path): + imgs_bits = [] + s_suffixes = set(suffixes) + reader = FileBasedDataReader(path) + for root, _, files in os.walk(path): + for file in files: + suffix = file.split('.') + if suffix[-1] in s_suffixes: + imgs_bits.append(reader.read(file)) + return [ImageDataset(bits) for bits in imgs_bits] + else: + reader = FileBasedDataReader() + bits = reader.read(path) + return [ImageDataset(bits)] diff --git a/magic_pdf/data/schemas.py b/magic_pdf/data/schemas.py new file mode 100644 index 00000000..3adb6760 --- /dev/null +++ b/magic_pdf/data/schemas.py @@ -0,0 +1,15 @@ + +from pydantic import BaseModel, Field + + +class S3Config(BaseModel): + bucket_name: str = Field(description='s3 bucket name', min_length=1) + access_key: str = Field(description='s3 access key', min_length=1) + secret_key: str = Field(description='s3 secret key', min_length=1) + endpoint_url: str = Field(description='s3 endpoint url', min_length=1) + addressing_style: str = Field(description='s3 addressing style', default='auto', min_length=1) + + +class PageInfo(BaseModel): + w: float = Field(description='the width of page') + h: float = Field(description='the height of page') diff --git a/magic_pdf/data/utils.py b/magic_pdf/data/utils.py new file mode 100644 index 00000000..dfe7abde --- /dev/null +++ b/magic_pdf/data/utils.py @@ -0,0 +1,32 @@ + +import fitz +import numpy as np + +from magic_pdf.utils.annotations import ImportPIL + + +@ImportPIL +def fitz_doc_to_image(doc, dpi=200) -> dict: + """Convert fitz.Document to image, Then convert the image to numpy array. + + Args: + doc (_type_): pymudoc page + dpi (int, optional): reset the dpi of dpi. Defaults to 200. + + Returns: + dict: {'img': numpy array, 'width': width, 'height': height } + """ + from PIL import Image + mat = fitz.Matrix(dpi / 72, dpi / 72) + pm = doc.get_pixmap(matrix=mat, alpha=False) + + # If the width or height exceeds 9000 after scaling, do not scale further. + if pm.width > 9000 or pm.height > 9000: + pm = doc.get_pixmap(matrix=fitz.Matrix(1, 1), alpha=False) + + img = Image.frombytes('RGB', (pm.width, pm.height), pm.samples) + img = np.array(img) + + img_dict = {'img': img, 'width': pm.width, 'height': pm.height} + + return img_dict diff --git a/magic_pdf/dict2md/ocr_mkcontent.py b/magic_pdf/dict2md/ocr_mkcontent.py index f438f6e6..91f8faf4 100644 --- a/magic_pdf/dict2md/ocr_mkcontent.py +++ b/magic_pdf/dict2md/ocr_mkcontent.py @@ -1,6 +1,5 @@ import re -import wordninja from loguru import logger from magic_pdf.libs.commons import join_path @@ -25,37 +24,6 @@ def __is_hyphen_at_line_end(line): return bool(re.search(r'[A-Za-z]+-\s*$', line)) -def split_long_words(text): - segments = text.split(' ') - for i in range(len(segments)): - words = re.findall(r'\w+|[^\w]', segments[i], re.UNICODE) - for j in range(len(words)): - if len(words[j]) > 10: - words[j] = ' '.join(wordninja.split(words[j])) - segments[i] = ''.join(words) - return ' '.join(segments) - - -def ocr_mk_mm_markdown_with_para(pdf_info_list: list, img_buket_path): - markdown = [] - for page_info in pdf_info_list: - paras_of_layout = page_info.get('para_blocks') - page_markdown = ocr_mk_markdown_with_para_core_v2( - paras_of_layout, 'mm', img_buket_path) - markdown.extend(page_markdown) - return '\n\n'.join(markdown) - - -def ocr_mk_nlp_markdown_with_para(pdf_info_dict: list): - markdown = [] - for page_info in pdf_info_dict: - paras_of_layout = page_info.get('para_blocks') - page_markdown = ocr_mk_markdown_with_para_core_v2( - paras_of_layout, 'nlp') - markdown.extend(page_markdown) - return '\n\n'.join(markdown) - - def ocr_mk_mm_markdown_with_para_and_pagination(pdf_info_dict: list, img_buket_path): markdown_with_para_and_pagination = [] @@ -68,69 +36,28 @@ def ocr_mk_mm_markdown_with_para_and_pagination(pdf_info_dict: list, paras_of_layout, 'mm', img_buket_path) markdown_with_para_and_pagination.append({ 'page_no': - page_no, + page_no, 'md_content': - '\n\n'.join(page_markdown) + '\n\n'.join(page_markdown) }) page_no += 1 return markdown_with_para_and_pagination -def ocr_mk_markdown_with_para_core(paras_of_layout, mode, img_buket_path=''): - page_markdown = [] - for paras in paras_of_layout: - for para in paras: - para_text = '' - for line in para: - for span in line['spans']: - span_type = span.get('type') - content = '' - language = '' - if span_type == ContentType.Text: - content = span['content'] - language = detect_lang(content) - if (language == 'en'): # 只对英文长词进行分词处理,中文分词会丢失文本 - content = ocr_escape_special_markdown_char( - split_long_words(content)) - else: - content = ocr_escape_special_markdown_char(content) - elif span_type == ContentType.InlineEquation: - content = f"${span['content']}$" - elif span_type == ContentType.InterlineEquation: - content = f"\n$$\n{span['content']}\n$$\n" - elif span_type in [ContentType.Image, ContentType.Table]: - if mode == 'mm': - content = f"\n![]({join_path(img_buket_path, span['image_path'])})\n" - elif mode == 'nlp': - pass - if content != '': - if language == 'en': # 英文语境下 content间需要空格分隔 - para_text += content + ' ' - else: # 中文语境下,content间不需要空格分隔 - para_text += content - if para_text.strip() == '': - continue - else: - page_markdown.append(para_text.strip() + ' ') - return page_markdown - - def ocr_mk_markdown_with_para_core_v2(paras_of_layout, mode, img_buket_path='', - parse_type="auto", - lang=None ): page_markdown = [] for para_block in paras_of_layout: para_text = '' para_type = para_block['type'] if para_type in [BlockType.Text, BlockType.List, BlockType.Index]: - para_text = merge_para_with_text(para_block, parse_type=parse_type, lang=lang) + para_text = merge_para_with_text(para_block) elif para_type == BlockType.Title: - para_text = f'# {merge_para_with_text(para_block, parse_type=parse_type, lang=lang)}' + para_text = f'# {merge_para_with_text(para_block)}' elif para_type == BlockType.InterlineEquation: - para_text = merge_para_with_text(para_block, parse_type=parse_type, lang=lang) + para_text = merge_para_with_text(para_block) elif para_type == BlockType.Image: if mode == 'nlp': continue @@ -143,17 +70,17 @@ def ocr_mk_markdown_with_para_core_v2(paras_of_layout, para_text += f"\n![]({join_path(img_buket_path, span['image_path'])}) \n" for block in para_block['blocks']: # 2nd.拼image_caption if block['type'] == BlockType.ImageCaption: - para_text += merge_para_with_text(block, parse_type=parse_type, lang=lang) - for block in para_block['blocks']: # 2nd.拼image_caption + para_text += merge_para_with_text(block) + ' \n' + for block in para_block['blocks']: # 3rd.拼image_footnote if block['type'] == BlockType.ImageFootnote: - para_text += merge_para_with_text(block, parse_type=parse_type, lang=lang) + para_text += merge_para_with_text(block) + ' \n' elif para_type == BlockType.Table: if mode == 'nlp': continue elif mode == 'mm': for block in para_block['blocks']: # 1st.拼table_caption if block['type'] == BlockType.TableCaption: - para_text += merge_para_with_text(block, parse_type=parse_type, lang=lang) + para_text += merge_para_with_text(block) + ' \n' for block in para_block['blocks']: # 2nd.拼table_body if block['type'] == BlockType.TableBody: for line in block['lines']: @@ -168,7 +95,7 @@ def ocr_mk_markdown_with_para_core_v2(paras_of_layout, para_text += f"\n![]({join_path(img_buket_path, span['image_path'])}) \n" for block in para_block['blocks']: # 3rd.拼table_footnote if block['type'] == BlockType.TableFootnote: - para_text += merge_para_with_text(block, parse_type=parse_type, lang=lang) + para_text += merge_para_with_text(block) + ' \n' if para_text.strip() == '': continue @@ -191,7 +118,7 @@ def detect_language(text): return 'empty' -def merge_para_with_text(para_block, parse_type="auto", lang=None): +def merge_para_with_text(para_block): para_text = '' for i, line in enumerate(para_block['lines']): @@ -207,21 +134,11 @@ def merge_para_with_text(para_block, parse_type="auto", lang=None): if line_text != '': line_lang = detect_lang(line_text) for span in line['spans']: + span_type = span['type'] content = '' if span_type == ContentType.Text: - content = span['content'] - # language = detect_lang(content) - language = detect_language(content) - # 判断是否小语种 - if lang is not None and lang != 'en': - content = ocr_escape_special_markdown_char(content) - else: # 非小语种逻辑 - if language == 'en' and parse_type == 'ocr': # 只对英文长词进行分词处理,中文分词会丢失文本 - content = ocr_escape_special_markdown_char( - split_long_words(content)) - else: - content = ocr_escape_special_markdown_char(content) + content = ocr_escape_special_markdown_char(span['content']) elif span_type == ContentType.InlineEquation: content = f" ${span['content']}$ " elif span_type == ContentType.InterlineEquation: @@ -242,74 +159,39 @@ def merge_para_with_text(para_block, parse_type="auto", lang=None): return para_text -def para_to_standard_format(para, img_buket_path): - para_content = {} - if len(para) == 1: - para_content = line_to_standard_format(para[0], img_buket_path) - elif len(para) > 1: - para_text = '' - inline_equation_num = 0 - for line in para: - for span in line['spans']: - language = '' - span_type = span.get('type') - content = '' - if span_type == ContentType.Text: - content = span['content'] - language = detect_lang(content) - if language == 'en': # 只对英文长词进行分词处理,中文分词会丢失文本 - content = ocr_escape_special_markdown_char( - split_long_words(content)) - else: - content = ocr_escape_special_markdown_char(content) - elif span_type == ContentType.InlineEquation: - content = f"${span['content']}$" - inline_equation_num += 1 - if language == 'en': # 英文语境下 content间需要空格分隔 - para_text += content + ' ' - else: # 中文语境下,content间不需要空格分隔 - para_text += content - para_content = { - 'type': 'text', - 'text': para_text, - 'inline_equation_num': inline_equation_num, - } - return para_content - - -def para_to_standard_format_v2(para_block, img_buket_path, page_idx, parse_type="auto", lang=None, drop_reason=None): +def para_to_standard_format_v2(para_block, img_buket_path, page_idx, drop_reason=None): para_type = para_block['type'] para_content = {} - if para_type == BlockType.Text: + if para_type in [BlockType.Text, BlockType.List, BlockType.Index]: para_content = { 'type': 'text', - 'text': merge_para_with_text(para_block, parse_type=parse_type, lang=lang), + 'text': merge_para_with_text(para_block), } elif para_type == BlockType.Title: para_content = { 'type': 'text', - 'text': merge_para_with_text(para_block, parse_type=parse_type, lang=lang), + 'text': merge_para_with_text(para_block), 'text_level': 1, } elif para_type == BlockType.InterlineEquation: para_content = { 'type': 'equation', - 'text': merge_para_with_text(para_block, parse_type=parse_type, lang=lang), + 'text': merge_para_with_text(para_block), 'text_format': 'latex', } elif para_type == BlockType.Image: - para_content = {'type': 'image'} + para_content = {'type': 'image', 'img_caption': [], 'img_footnote': []} for block in para_block['blocks']: if block['type'] == BlockType.ImageBody: para_content['img_path'] = join_path( img_buket_path, block['lines'][0]['spans'][0]['image_path']) if block['type'] == BlockType.ImageCaption: - para_content['img_caption'] = merge_para_with_text(block, parse_type=parse_type, lang=lang) + para_content['img_caption'].append(merge_para_with_text(block)) if block['type'] == BlockType.ImageFootnote: - para_content['img_footnote'] = merge_para_with_text(block, parse_type=parse_type, lang=lang) + para_content['img_footnote'].append(merge_para_with_text(block)) elif para_type == BlockType.Table: - para_content = {'type': 'table'} + para_content = {'type': 'table', 'table_caption': [], 'table_footnote': []} for block in para_block['blocks']: if block['type'] == BlockType.TableBody: if block["lines"][0]["spans"][0].get('latex', ''): @@ -318,9 +200,9 @@ def para_to_standard_format_v2(para_block, img_buket_path, page_idx, parse_type= para_content['table_body'] = f"\n\n{block['lines'][0]['spans'][0]['html']}\n\n" para_content['img_path'] = join_path(img_buket_path, block["lines"][0]["spans"][0]['image_path']) if block['type'] == BlockType.TableCaption: - para_content['table_caption'] = merge_para_with_text(block, parse_type=parse_type, lang=lang) + para_content['table_caption'].append(merge_para_with_text(block)) if block['type'] == BlockType.TableFootnote: - para_content['table_footnote'] = merge_para_with_text(block, parse_type=parse_type, lang=lang) + para_content['table_footnote'].append(merge_para_with_text(block)) para_content['page_idx'] = page_idx @@ -330,88 +212,11 @@ def para_to_standard_format_v2(para_block, img_buket_path, page_idx, parse_type= return para_content -def make_standard_format_with_para(pdf_info_dict: list, img_buket_path: str): - content_list = [] - for page_info in pdf_info_dict: - paras_of_layout = page_info.get('para_blocks') - if not paras_of_layout: - continue - for para_block in paras_of_layout: - para_content = para_to_standard_format_v2(para_block, - img_buket_path) - content_list.append(para_content) - return content_list - - -def line_to_standard_format(line, img_buket_path): - line_text = '' - inline_equation_num = 0 - for span in line['spans']: - if not span.get('content'): - if not span.get('image_path'): - continue - else: - if span['type'] == ContentType.Image: - content = { - 'type': 'image', - 'img_path': join_path(img_buket_path, - span['image_path']), - } - return content - elif span['type'] == ContentType.Table: - content = { - 'type': 'table', - 'img_path': join_path(img_buket_path, - span['image_path']), - } - return content - else: - if span['type'] == ContentType.InterlineEquation: - interline_equation = span['content'] - content = { - 'type': 'equation', - 'latex': f'$$\n{interline_equation}\n$$' - } - return content - elif span['type'] == ContentType.InlineEquation: - inline_equation = span['content'] - line_text += f'${inline_equation}$' - inline_equation_num += 1 - elif span['type'] == ContentType.Text: - text_content = ocr_escape_special_markdown_char( - span['content']) # 转义特殊符号 - line_text += text_content - content = { - 'type': 'text', - 'text': line_text, - 'inline_equation_num': inline_equation_num, - } - return content - - -def ocr_mk_mm_standard_format(pdf_info_dict: list): - """content_list type string - image/text/table/equation(行间的单独拿出来,行内的和text合并) latex string - latex文本字段。 text string 纯文本格式的文本数据。 md string - markdown格式的文本数据。 img_path string s3://full/path/to/img.jpg.""" - content_list = [] - for page_info in pdf_info_dict: - blocks = page_info.get('preproc_blocks') - if not blocks: - continue - for block in blocks: - for line in block['lines']: - content = line_to_standard_format(line) - content_list.append(content) - return content_list - - def union_make(pdf_info_dict: list, make_mode: str, drop_mode: str, img_buket_path: str = '', - parse_type: str = "auto", - lang=None): + ): output_content = [] for page_info in pdf_info_dict: drop_reason_flag = False @@ -438,20 +243,20 @@ def union_make(pdf_info_dict: list, continue if make_mode == MakeMode.MM_MD: page_markdown = ocr_mk_markdown_with_para_core_v2( - paras_of_layout, 'mm', img_buket_path, parse_type=parse_type, lang=lang) + paras_of_layout, 'mm', img_buket_path) output_content.extend(page_markdown) elif make_mode == MakeMode.NLP_MD: page_markdown = ocr_mk_markdown_with_para_core_v2( - paras_of_layout, 'nlp', parse_type=parse_type, lang=lang) + paras_of_layout, 'nlp') output_content.extend(page_markdown) elif make_mode == MakeMode.STANDARD_FORMAT: for para_block in paras_of_layout: if drop_reason_flag: para_content = para_to_standard_format_v2( - para_block, img_buket_path, page_idx, parse_type=parse_type, lang=lang, drop_reason=drop_reason) + para_block, img_buket_path, page_idx) else: para_content = para_to_standard_format_v2( - para_block, img_buket_path, page_idx, parse_type=parse_type, lang=lang) + para_block, img_buket_path, page_idx) output_content.append(para_content) if make_mode in [MakeMode.MM_MD, MakeMode.NLP_MD]: return '\n\n'.join(output_content) diff --git a/magic_pdf/libs/Constants.py b/magic_pdf/libs/Constants.py index e6fa4b78..0799f6fd 100644 --- a/magic_pdf/libs/Constants.py +++ b/magic_pdf/libs/Constants.py @@ -10,18 +10,12 @@ block维度自定义字段 # block中lines是否被删除 LINES_DELETED = "lines_deleted" -# struct eqtable -STRUCT_EQTABLE = "struct_eqtable" - # table recognition max time default value TABLE_MAX_TIME_VALUE = 400 # pp_table_result_max_length TABLE_MAX_LEN = 480 -# pp table structure algorithm -TABLE_MASTER = "TableMaster" - # table master structure dict TABLE_MASTER_DICT = "table_master_structure_dict.txt" @@ -44,3 +38,16 @@ PP_REC_DIRECTORY = ".paddleocr/whl/rec/ch/ch_PP-OCRv4_rec_infer" PP_DET_DIRECTORY = ".paddleocr/whl/det/ch/ch_PP-OCRv4_det_infer" +class MODEL_NAME: + # pp table structure algorithm + TABLE_MASTER = "tablemaster" + # struct eqtable + STRUCT_EQTABLE = "struct_eqtable" + + DocLayout_YOLO = "doclayout_yolo" + + LAYOUTLMv3 = "layoutlmv3" + + YOLO_V8_MFD = "yolo_v8_mfd" + + UniMerNet_v2_Small = "unimernet_small" \ No newline at end of file diff --git a/magic_pdf/libs/boxbase.py b/magic_pdf/libs/boxbase.py index 0472328f..52779a22 100644 --- a/magic_pdf/libs/boxbase.py +++ b/magic_pdf/libs/boxbase.py @@ -445,3 +445,38 @@ def get_overlap_area(bbox1, bbox2): # The area of overlap area return (x_right - x_left) * (y_bottom - y_top) + + +def calculate_vertical_projection_overlap_ratio(block1, block2): + """ + Calculate the proportion of the x-axis covered by the vertical projection of two blocks. + + Args: + block1 (tuple): Coordinates of the first block (x0, y0, x1, y1). + block2 (tuple): Coordinates of the second block (x0, y0, x1, y1). + + Returns: + float: The proportion of the x-axis covered by the vertical projection of the two blocks. + """ + x0_1, _, x1_1, _ = block1 + x0_2, _, x1_2, _ = block2 + + # Calculate the intersection of the x-coordinates + x_left = max(x0_1, x0_2) + x_right = min(x1_1, x1_2) + + if x_right < x_left: + return 0.0 + + # Length of the intersection + intersection_length = x_right - x_left + + # Length of the x-axis projection of the first block + block1_length = x1_1 - x0_1 + + if block1_length == 0: + return 0.0 + + # Proportion of the x-axis covered by the intersection + # logger.info(f"intersection_length: {intersection_length}, block1_length: {block1_length}") + return intersection_length / block1_length diff --git a/magic_pdf/libs/config_reader.py b/magic_pdf/libs/config_reader.py index 9b4b7d8b..5e1a300d 100644 --- a/magic_pdf/libs/config_reader.py +++ b/magic_pdf/libs/config_reader.py @@ -1,46 +1,44 @@ -""" -根据bucket的名字返回对应的s3 AK, SK,endpoint三元组 - -""" +"""根据bucket的名字返回对应的s3 AK, SK,endpoint三元组.""" import json import os from loguru import logger +from magic_pdf.libs.Constants import MODEL_NAME from magic_pdf.libs.commons import parse_bucket_key # 定义配置文件名常量 -CONFIG_FILE_NAME = "magic-pdf.json" +CONFIG_FILE_NAME = os.getenv('MINERU_TOOLS_CONFIG_JSON', 'magic-pdf.json') def read_config(): - home_dir = os.path.expanduser("~") - - config_file = os.path.join(home_dir, CONFIG_FILE_NAME) + if os.path.isabs(CONFIG_FILE_NAME): + config_file = CONFIG_FILE_NAME + else: + home_dir = os.path.expanduser('~') + config_file = os.path.join(home_dir, CONFIG_FILE_NAME) if not os.path.exists(config_file): - raise FileNotFoundError(f"{config_file} not found") + raise FileNotFoundError(f'{config_file} not found') - with open(config_file, "r", encoding="utf-8") as f: + with open(config_file, 'r', encoding='utf-8') as f: config = json.load(f) return config def get_s3_config(bucket_name: str): - """ - ~/magic-pdf.json 读出来 - """ + """~/magic-pdf.json 读出来.""" config = read_config() - bucket_info = config.get("bucket_info") + bucket_info = config.get('bucket_info') if bucket_name not in bucket_info: - access_key, secret_key, storage_endpoint = bucket_info["[default]"] + access_key, secret_key, storage_endpoint = bucket_info['[default]'] else: access_key, secret_key, storage_endpoint = bucket_info[bucket_name] if access_key is None or secret_key is None or storage_endpoint is None: - raise Exception(f"ak, sk or endpoint not found in {CONFIG_FILE_NAME}") + raise Exception(f'ak, sk or endpoint not found in {CONFIG_FILE_NAME}') # logger.info(f"get_s3_config: ak={access_key}, sk={secret_key}, endpoint={storage_endpoint}") @@ -49,7 +47,7 @@ def get_s3_config(bucket_name: str): def get_s3_config_dict(path: str): access_key, secret_key, storage_endpoint = get_s3_config(get_bucket_name(path)) - return {"ak": access_key, "sk": secret_key, "endpoint": storage_endpoint} + return {'ak': access_key, 'sk': secret_key, 'endpoint': storage_endpoint} def get_bucket_name(path): @@ -59,20 +57,20 @@ def get_bucket_name(path): def get_local_models_dir(): config = read_config() - models_dir = config.get("models-dir") + models_dir = config.get('models-dir') if models_dir is None: logger.warning(f"'models-dir' not found in {CONFIG_FILE_NAME}, use '/tmp/models' as default") - return "/tmp/models" + return '/tmp/models' else: return models_dir def get_local_layoutreader_model_dir(): config = read_config() - layoutreader_model_dir = config.get("layoutreader-model-dir") + layoutreader_model_dir = config.get('layoutreader-model-dir') if layoutreader_model_dir is None or not os.path.exists(layoutreader_model_dir): - home_dir = os.path.expanduser("~") - layoutreader_at_modelscope_dir_path = os.path.join(home_dir, ".cache/modelscope/hub/ppaanngggg/layoutreader") + home_dir = os.path.expanduser('~') + layoutreader_at_modelscope_dir_path = os.path.join(home_dir, '.cache/modelscope/hub/ppaanngggg/layoutreader') logger.warning(f"'layoutreader-model-dir' not exists, use {layoutreader_at_modelscope_dir_path} as default") return layoutreader_at_modelscope_dir_path else: @@ -81,23 +79,43 @@ def get_local_layoutreader_model_dir(): def get_device(): config = read_config() - device = config.get("device-mode") + device = config.get('device-mode') if device is None: logger.warning(f"'device-mode' not found in {CONFIG_FILE_NAME}, use 'cpu' as default") - return "cpu" + return 'cpu' else: return device def get_table_recog_config(): config = read_config() - table_config = config.get("table-config") + table_config = config.get('table-config') if table_config is None: logger.warning(f"'table-config' not found in {CONFIG_FILE_NAME}, use 'False' as default") - return json.loads('{"is_table_recog_enable": false, "max_time": 400}') + return json.loads(f'{{"model": "{MODEL_NAME.TABLE_MASTER}","enable": false, "max_time": 400}}') else: return table_config +def get_layout_config(): + config = read_config() + layout_config = config.get("layout-config") + if layout_config is None: + logger.warning(f"'layout-config' not found in {CONFIG_FILE_NAME}, use '{MODEL_NAME.LAYOUTLMv3}' as default") + return json.loads(f'{{"model": "{MODEL_NAME.LAYOUTLMv3}"}}') + else: + return layout_config + + +def get_formula_config(): + config = read_config() + formula_config = config.get("formula-config") + if formula_config is None: + logger.warning(f"'formula-config' not found in {CONFIG_FILE_NAME}, use 'True' as default") + return json.loads(f'{{"mfd_model": "{MODEL_NAME.YOLO_V8_MFD}","mfr_model": "{MODEL_NAME.UniMerNet_v2_Small}","enable": true}}') + else: + return formula_config + + if __name__ == "__main__": ak, sk, endpoint = get_s3_config("llm-raw") diff --git a/magic_pdf/libs/draw_bbox.py b/magic_pdf/libs/draw_bbox.py index 550a4cec..9703e131 100644 --- a/magic_pdf/libs/draw_bbox.py +++ b/magic_pdf/libs/draw_bbox.py @@ -1,3 +1,4 @@ +from magic_pdf.data.dataset import PymuDocDataset from magic_pdf.libs.commons import fitz # PyMuPDF from magic_pdf.libs.Constants import CROSS_PAGE from magic_pdf.libs.ocr_content_type import BlockType, CategoryId, ContentType @@ -62,7 +63,7 @@ def draw_bbox_with_number(i, bbox_list, page, rgb_config, fill_config, draw_bbox overlay=True, ) # Draw the rectangle page.insert_text( - (x1+2, y0 + 10), str(j + 1), fontsize=10, color=new_rgb + (x1 + 2, y0 + 10), str(j + 1), fontsize=10, color=new_rgb ) # Insert the index in the top left corner of the rectangle @@ -86,7 +87,7 @@ def draw_layout_bbox(pdf_info, pdf_bytes, out_path, filename): texts = [] interequations = [] lists = [] - indexs = [] + indices = [] for dropped_bbox in page['discarded_blocks']: page_dropped_list.append(dropped_bbox['bbox']) @@ -122,7 +123,7 @@ def draw_layout_bbox(pdf_info, pdf_bytes, out_path, filename): elif block['type'] == BlockType.List: lists.append(bbox) elif block['type'] == BlockType.Index: - indexs.append(bbox) + indices.append(bbox) tables_list.append(tables) tables_body_list.append(tables_body) @@ -136,45 +137,61 @@ def draw_layout_bbox(pdf_info, pdf_bytes, out_path, filename): texts_list.append(texts) interequations_list.append(interequations) lists_list.append(lists) - indexs_list.append(indexs) + indexs_list.append(indices) layout_bbox_list = [] + table_type_order = { + 'table_caption': 1, + 'table_body': 2, + 'table_footnote': 3 + } for page in pdf_info: page_block_list = [] for block in page['para_blocks']: - bbox = block['bbox'] - page_block_list.append(bbox) + if block['type'] in [ + BlockType.Text, + BlockType.Title, + BlockType.InterlineEquation, + BlockType.List, + BlockType.Index, + ]: + bbox = block['bbox'] + page_block_list.append(bbox) + elif block['type'] in [BlockType.Image]: + for sub_block in block['blocks']: + bbox = sub_block['bbox'] + page_block_list.append(bbox) + elif block['type'] in [BlockType.Table]: + sorted_blocks = sorted(block['blocks'], key=lambda x: table_type_order[x['type']]) + for sub_block in sorted_blocks: + bbox = sub_block['bbox'] + page_block_list.append(bbox) + layout_bbox_list.append(page_block_list) pdf_docs = fitz.open('pdf', pdf_bytes) for i, page in enumerate(pdf_docs): - draw_bbox_without_number(i, dropped_bbox_list, page, [158, 158, 158], - True) - draw_bbox_without_number(i, tables_list, page, [153, 153, 0], - True) # color ! - draw_bbox_without_number(i, tables_body_list, page, [204, 204, 0], - True) - draw_bbox_without_number(i, tables_caption_list, page, [255, 255, 102], - True) - draw_bbox_without_number(i, tables_footnote_list, page, - [229, 255, 204], True) - draw_bbox_without_number(i, imgs_list, page, [51, 102, 0], True) + draw_bbox_without_number(i, dropped_bbox_list, page, [158, 158, 158], True) + # draw_bbox_without_number(i, tables_list, page, [153, 153, 0], True) # color ! + draw_bbox_without_number(i, tables_body_list, page, [204, 204, 0], True) + draw_bbox_without_number(i, tables_caption_list, page, [255, 255, 102], True) + draw_bbox_without_number(i, tables_footnote_list, page, [229, 255, 204], True) + # draw_bbox_without_number(i, imgs_list, page, [51, 102, 0], True) draw_bbox_without_number(i, imgs_body_list, page, [153, 255, 51], True) - draw_bbox_without_number(i, imgs_caption_list, page, [102, 178, 255], - True) - draw_bbox_without_number(i, imgs_footnote_list, page, [255, 178, 102], - True), + draw_bbox_without_number(i, imgs_caption_list, page, [102, 178, 255], True) + draw_bbox_without_number(i, imgs_footnote_list, page, [255, 178, 102], True), draw_bbox_without_number(i, titles_list, page, [102, 102, 255], True) draw_bbox_without_number(i, texts_list, page, [153, 0, 76], True) - draw_bbox_without_number(i, interequations_list, page, [0, 255, 0], - True) + draw_bbox_without_number(i, interequations_list, page, [0, 255, 0], True) draw_bbox_without_number(i, lists_list, page, [40, 169, 92], True) draw_bbox_without_number(i, indexs_list, page, [40, 169, 92], True) - draw_bbox_with_number(i, layout_bbox_list, page, [255, 0, 0], False, draw_bbox=False) + draw_bbox_with_number( + i, layout_bbox_list, page, [255, 0, 0], False, draw_bbox=False + ) # Save the PDF pdf_docs.save(f'{out_path}/{filename}_layout.pdf') @@ -237,6 +254,8 @@ def draw_span_bbox(pdf_info, pdf_bytes, out_path, filename): BlockType.Text, BlockType.Title, BlockType.InterlineEquation, + BlockType.List, + BlockType.Index, ]: for line in block['lines']: for span in line['spans']: @@ -273,7 +292,7 @@ def draw_model_bbox(model_list: list, pdf_bytes, out_path, filename): texts_list = [] interequations_list = [] pdf_docs = fitz.open('pdf', pdf_bytes) - magic_model = MagicModel(model_list, pdf_docs) + magic_model = MagicModel(model_list, PymuDocDataset(pdf_bytes)) for i in range(len(model_list)): page_dropped_list = [] tables_body, tables_caption, tables_footnote = [], [], [] @@ -299,8 +318,7 @@ def draw_model_bbox(model_list: list, pdf_bytes, out_path, filename): imgs_body.append(bbox) elif layout_det['category_id'] == CategoryId.ImageCaption: imgs_caption.append(bbox) - elif layout_det[ - 'category_id'] == CategoryId.InterlineEquation_YOLO: + elif layout_det['category_id'] == CategoryId.InterlineEquation_YOLO: interequations.append(bbox) elif layout_det['category_id'] == CategoryId.Abandon: page_dropped_list.append(bbox) @@ -319,18 +337,15 @@ def draw_model_bbox(model_list: list, pdf_bytes, out_path, filename): imgs_footnote_list.append(imgs_footnote) for i, page in enumerate(pdf_docs): - draw_bbox_with_number(i, dropped_bbox_list, page, [158, 158, 158], - True) # color ! + draw_bbox_with_number( + i, dropped_bbox_list, page, [158, 158, 158], True + ) # color ! draw_bbox_with_number(i, tables_body_list, page, [204, 204, 0], True) - draw_bbox_with_number(i, tables_caption_list, page, [255, 255, 102], - True) - draw_bbox_with_number(i, tables_footnote_list, page, [229, 255, 204], - True) + draw_bbox_with_number(i, tables_caption_list, page, [255, 255, 102], True) + draw_bbox_with_number(i, tables_footnote_list, page, [229, 255, 204], True) draw_bbox_with_number(i, imgs_body_list, page, [153, 255, 51], True) - draw_bbox_with_number(i, imgs_caption_list, page, [102, 178, 255], - True) - draw_bbox_with_number(i, imgs_footnote_list, page, [255, 178, 102], - True) + draw_bbox_with_number(i, imgs_caption_list, page, [102, 178, 255], True) + draw_bbox_with_number(i, imgs_footnote_list, page, [255, 178, 102], True) draw_bbox_with_number(i, titles_list, page, [102, 102, 255], True) draw_bbox_with_number(i, texts_list, page, [153, 0, 76], True) draw_bbox_with_number(i, interequations_list, page, [0, 255, 0], True) @@ -345,19 +360,23 @@ def draw_line_sort_bbox(pdf_info, pdf_bytes, out_path, filename): for page in pdf_info: page_line_list = [] for block in page['preproc_blocks']: - if block['type'] in ['text', 'title', 'interline_equation']: + if block['type'] in [BlockType.Text, BlockType.Title, BlockType.InterlineEquation]: for line in block['lines']: bbox = line['bbox'] index = line['index'] page_line_list.append({'index': index, 'bbox': bbox}) - if block['type'] in ['table', 'image']: - bbox = block['bbox'] - index = block['index'] - page_line_list.append({'index': index, 'bbox': bbox}) - # for line in block['lines']: - # bbox = line['bbox'] - # index = line['index'] - # page_line_list.append({'index': index, 'bbox': bbox}) + if block['type'] in [BlockType.Image, BlockType.Table]: + for sub_block in block['blocks']: + if sub_block['type'] in [BlockType.ImageBody, BlockType.TableBody]: + for line in sub_block['virtual_lines']: + bbox = line['bbox'] + index = line['index'] + page_line_list.append({'index': index, 'bbox': bbox}) + elif sub_block['type'] in [BlockType.ImageCaption, BlockType.TableCaption, BlockType.ImageFootnote, BlockType.TableFootnote]: + for line in sub_block['lines']: + bbox = line['bbox'] + index = line['index'] + page_line_list.append({'index': index, 'bbox': bbox}) sorted_bboxes = sorted(page_line_list, key=lambda x: x['index']) layout_bbox_list.append(sorted_bbox['bbox'] for sorted_bbox in sorted_bboxes) pdf_docs = fitz.open('pdf', pdf_bytes) diff --git a/magic_pdf/model/doc_analyze_by_custom_model.py b/magic_pdf/model/doc_analyze_by_custom_model.py index 3fbbea61..ee50d6eb 100644 --- a/magic_pdf/model/doc_analyze_by_custom_model.py +++ b/magic_pdf/model/doc_analyze_by_custom_model.py @@ -5,7 +5,8 @@ import numpy as np from loguru import logger from magic_pdf.libs.clean_memory import clean_memory -from magic_pdf.libs.config_reader import get_local_models_dir, get_device, get_table_recog_config +from magic_pdf.libs.config_reader import get_local_models_dir, get_device, get_table_recog_config, get_layout_config, \ + get_formula_config from magic_pdf.model.model_list import MODEL import magic_pdf.model as model_config @@ -68,14 +69,17 @@ class ModelSingleton: cls._instance = super().__new__(cls) return cls._instance - def get_model(self, ocr: bool, show_log: bool, lang=None): - key = (ocr, show_log, lang) + def get_model(self, ocr: bool, show_log: bool, lang=None, layout_model=None, formula_enable=None, table_enable=None): + key = (ocr, show_log, lang, layout_model, formula_enable, table_enable) if key not in self._models: - self._models[key] = custom_model_init(ocr=ocr, show_log=show_log, lang=lang) + self._models[key] = custom_model_init(ocr=ocr, show_log=show_log, lang=lang, layout_model=layout_model, + formula_enable=formula_enable, table_enable=table_enable) return self._models[key] -def custom_model_init(ocr: bool = False, show_log: bool = False, lang=None): +def custom_model_init(ocr: bool = False, show_log: bool = False, lang=None, + layout_model=None, formula_enable=None, table_enable=None): + model = None if model_config.__model_mode__ == "lite": @@ -95,14 +99,30 @@ def custom_model_init(ocr: bool = False, show_log: bool = False, lang=None): # 从配置文件读取model-dir和device local_models_dir = get_local_models_dir() device = get_device() + + layout_config = get_layout_config() + if layout_model is not None: + layout_config["model"] = layout_model + + formula_config = get_formula_config() + if formula_enable is not None: + formula_config["enable"] = formula_enable + table_config = get_table_recog_config() - model_input = {"ocr": ocr, - "show_log": show_log, - "models_dir": local_models_dir, - "device": device, - "table_config": table_config, - "lang": lang, - } + if table_enable is not None: + table_config["enable"] = table_enable + + model_input = { + "ocr": ocr, + "show_log": show_log, + "models_dir": local_models_dir, + "device": device, + "table_config": table_config, + "layout_config": layout_config, + "formula_config": formula_config, + "lang": lang, + } + custom_model = CustomPEKModel(**model_input) else: logger.error("Not allow model_name!") @@ -117,10 +137,14 @@ def custom_model_init(ocr: bool = False, show_log: bool = False, lang=None): def doc_analyze(pdf_bytes: bytes, ocr: bool = False, show_log: bool = False, - start_page_id=0, end_page_id=None, lang=None): + start_page_id=0, end_page_id=None, lang=None, + layout_model=None, formula_enable=None, table_enable=None): + + if lang == "": + lang = None model_manager = ModelSingleton() - custom_model = model_manager.get_model(ocr, show_log, lang) + custom_model = model_manager.get_model(ocr, show_log, lang, layout_model, formula_enable, table_enable) with fitz.open("pdf", pdf_bytes) as doc: pdf_page_num = doc.page_count diff --git a/magic_pdf/model/magic_model.py b/magic_pdf/model/magic_model.py index 79feecae..62bbeb15 100644 --- a/magic_pdf/model/magic_model.py +++ b/magic_pdf/model/magic_model.py @@ -1,5 +1,6 @@ import json +from magic_pdf.data.dataset import Dataset from magic_pdf.libs.boxbase import (_is_in, _is_part_overlap, bbox_distance, bbox_relative_pos, box_area, calculate_iou, calculate_overlap_area_in_bbox1_area_ratio, @@ -9,6 +10,7 @@ from magic_pdf.libs.coordinate_transform import get_scale_ratio from magic_pdf.libs.local_math import float_gt from magic_pdf.libs.ModelBlockTypeEnum import ModelBlockTypeEnum from magic_pdf.libs.ocr_content_type import CategoryId, ContentType +from magic_pdf.pre_proc.remove_bbox_overlap import _remove_overlap_between_bbox from magic_pdf.rw.AbsReaderWriter import AbsReaderWriter from magic_pdf.rw.DiskReaderWriter import DiskReaderWriter @@ -24,7 +26,7 @@ class MagicModel: need_remove_list = [] page_no = model_page_info['page_info']['page_no'] horizontal_scale_ratio, vertical_scale_ratio = get_scale_ratio( - model_page_info, self.__docs[page_no] + model_page_info, self.__docs.get_page(page_no) ) layout_dets = model_page_info['layout_dets'] for layout_det in layout_dets: @@ -99,7 +101,7 @@ class MagicModel: for need_remove in need_remove_list: layout_dets.remove(need_remove) - def __init__(self, model_list: list, docs: fitz.Document): + def __init__(self, model_list: list, docs: Dataset): self.__model_list = model_list self.__docs = docs """为所有模型数据添加bbox信息(缩放,poly->bbox)""" @@ -123,7 +125,7 @@ class MagicModel: l1 = bbox1[2] - bbox1[0] l2 = bbox2[2] - bbox2[0] - if l2 > l1 and (l2 - l1) / l1 > 0.5: + if l2 > l1 and (l2 - l1) / l1 > 0.3: return float('inf') return bbox_distance(bbox1, bbox2) @@ -213,9 +215,8 @@ class MagicModel: 筛选出所有和 merged bbox 有 overlap 且 overlap 面积大于 object 的面积的 subjects。 再求出筛选出的 subjects 和 object 的最短距离 """ - def search_overlap_between_boxes( - subject_idx, object_idx - ): + + def search_overlap_between_boxes(subject_idx, object_idx): idxes = [subject_idx, object_idx] x0s = [all_bboxes[idx]['bbox'][0] for idx in idxes] y0s = [all_bboxes[idx]['bbox'][1] for idx in idxes] @@ -243,9 +244,9 @@ class MagicModel: for other_object in other_objects: ratio = max( ratio, - get_overlap_area( - merged_bbox, other_object['bbox'] - ) * 1.0 / box_area(all_bboxes[object_idx]['bbox']) + get_overlap_area(merged_bbox, other_object['bbox']) + * 1.0 + / box_area(all_bboxes[object_idx]['bbox']), ) if ratio >= MERGE_BOX_OVERLAP_AREA_RATIO: break @@ -363,12 +364,17 @@ class MagicModel: if all_bboxes[j]['category_id'] == subject_category_id: subject_idx, object_idx = j, i - if search_overlap_between_boxes(subject_idx, object_idx) >= MERGE_BOX_OVERLAP_AREA_RATIO: + if ( + search_overlap_between_boxes(subject_idx, object_idx) + >= MERGE_BOX_OVERLAP_AREA_RATIO + ): dis[i][j] = float('inf') dis[j][i] = dis[i][j] continue - dis[i][j] = self._bbox_distance(all_bboxes[subject_idx]['bbox'], all_bboxes[object_idx]['bbox']) + dis[i][j] = self._bbox_distance( + all_bboxes[subject_idx]['bbox'], all_bboxes[object_idx]['bbox'] + ) dis[j][i] = dis[i][j] used = set() @@ -584,6 +590,245 @@ class MagicModel: with_caption_subject.add(j) return ret, total_subject_object_dis + def __tie_up_category_by_distance_v2( + self, page_no, subject_category_id, object_category_id + ): + + AXIS_MULPLICITY = 0.5 + subjects = self.__reduct_overlap( + list( + map( + lambda x: {'bbox': x['bbox'], 'score': x['score']}, + filter( + lambda x: x['category_id'] == subject_category_id, + self.__model_list[page_no]['layout_dets'], + ), + ) + ) + ) + + objects = self.__reduct_overlap( + list( + map( + lambda x: {'bbox': x['bbox'], 'score': x['score']}, + filter( + lambda x: x['category_id'] == object_category_id, + self.__model_list[page_no]['layout_dets'], + ), + ) + ) + ) + M = len(objects) + + subjects.sort(key=lambda x: x['bbox'][0] ** 2 + x['bbox'][1] ** 2) + objects.sort(key=lambda x: x['bbox'][0] ** 2 + x['bbox'][1] ** 2) + + sub_obj_map_h = {i: [] for i in range(len(subjects))} + + dis_by_directions = { + 'top': [[-1, float('inf')]] * M, + 'bottom': [[-1, float('inf')]] * M, + 'left': [[-1, float('inf')]] * M, + 'right': [[-1, float('inf')]] * M, + } + + for i, obj in enumerate(objects): + l_x_axis, l_y_axis = ( + obj['bbox'][2] - obj['bbox'][0], + obj['bbox'][3] - obj['bbox'][1], + ) + axis_unit = min(l_x_axis, l_y_axis) + for j, sub in enumerate(subjects): + + bbox1, bbox2, _ = _remove_overlap_between_bbox( + objects[i]['bbox'], subjects[j]['bbox'] + ) + left, right, bottom, top = bbox_relative_pos(bbox1, bbox2) + flags = [left, right, bottom, top] + if sum([1 if v else 0 for v in flags]) > 1: + continue + + if left: + if dis_by_directions['left'][i][1] > bbox_distance( + obj['bbox'], sub['bbox'] + ): + dis_by_directions['left'][i] = [ + j, + bbox_distance(obj['bbox'], sub['bbox']), + ] + if right: + if dis_by_directions['right'][i][1] > bbox_distance( + obj['bbox'], sub['bbox'] + ): + dis_by_directions['right'][i] = [ + j, + bbox_distance(obj['bbox'], sub['bbox']), + ] + if bottom: + if dis_by_directions['bottom'][i][1] > bbox_distance( + obj['bbox'], sub['bbox'] + ): + dis_by_directions['bottom'][i] = [ + j, + bbox_distance(obj['bbox'], sub['bbox']), + ] + if top: + if dis_by_directions['top'][i][1] > bbox_distance( + obj['bbox'], sub['bbox'] + ): + dis_by_directions['top'][i] = [ + j, + bbox_distance(obj['bbox'], sub['bbox']), + ] + if dis_by_directions['left'][i][1] != float('inf') or dis_by_directions[ + 'right' + ][i][1] != float('inf'): + if dis_by_directions['left'][i][1] != float( + 'inf' + ) and dis_by_directions['right'][i][1] != float('inf'): + if AXIS_MULPLICITY * axis_unit >= abs( + dis_by_directions['left'][i][1] + - dis_by_directions['right'][i][1] + ): + left_sub_bbox = subjects[dis_by_directions['left'][i][0]][ + 'bbox' + ] + right_sub_bbox = subjects[dis_by_directions['right'][i][0]][ + 'bbox' + ] + + left_sub_bbox_y_axis = left_sub_bbox[3] - left_sub_bbox[1] + right_sub_bbox_y_axis = right_sub_bbox[3] - right_sub_bbox[1] + + if ( + abs(left_sub_bbox_y_axis - l_y_axis) + + dis_by_directions['left'][i][0] + > abs(right_sub_bbox_y_axis - l_y_axis) + + dis_by_directions['right'][i][0] + ): + left_or_right = dis_by_directions['right'][i] + else: + left_or_right = dis_by_directions['left'][i] + else: + left_or_right = dis_by_directions['left'][i] + if left_or_right[1] > dis_by_directions['right'][i][1]: + left_or_right = dis_by_directions['right'][i] + else: + left_or_right = dis_by_directions['left'][i] + if left_or_right[1] == float('inf'): + left_or_right = dis_by_directions['right'][i] + else: + left_or_right = [-1, float('inf')] + + if dis_by_directions['top'][i][1] != float('inf') or dis_by_directions[ + 'bottom' + ][i][1] != float('inf'): + if dis_by_directions['top'][i][1] != float('inf') and dis_by_directions[ + 'bottom' + ][i][1] != float('inf'): + if AXIS_MULPLICITY * axis_unit >= abs( + dis_by_directions['top'][i][1] + - dis_by_directions['bottom'][i][1] + ): + top_bottom = subjects[dis_by_directions['bottom'][i][0]]['bbox'] + bottom_top = subjects[dis_by_directions['top'][i][0]]['bbox'] + + top_bottom_x_axis = top_bottom[2] - top_bottom[0] + bottom_top_x_axis = bottom_top[2] - bottom_top[0] + if abs(top_bottom_x_axis - l_x_axis) + dis_by_directions['bottom'][i][1] > abs( + bottom_top_x_axis - l_x_axis + ) + dis_by_directions['top'][i][1]: + top_or_bottom = dis_by_directions['top'][i] + else: + top_or_bottom = dis_by_directions['bottom'][i] + else: + top_or_bottom = dis_by_directions['top'][i] + if top_or_bottom[1] > dis_by_directions['bottom'][i][1]: + top_or_bottom = dis_by_directions['bottom'][i] + else: + top_or_bottom = dis_by_directions['top'][i] + if top_or_bottom[1] == float('inf'): + top_or_bottom = dis_by_directions['bottom'][i] + else: + top_or_bottom = [-1, float('inf')] + + if left_or_right[1] != float('inf') or top_or_bottom[1] != float('inf'): + if left_or_right[1] != float('inf') and top_or_bottom[1] != float( + 'inf' + ): + if AXIS_MULPLICITY * axis_unit >= abs( + left_or_right[1] - top_or_bottom[1] + ): + y_axis_bbox = subjects[left_or_right[0]]['bbox'] + x_axis_bbox = subjects[top_or_bottom[0]]['bbox'] + + if ( + abs((x_axis_bbox[2] - x_axis_bbox[0]) - l_x_axis) / l_x_axis + > abs((y_axis_bbox[3] - y_axis_bbox[1]) - l_y_axis) + / l_y_axis + ): + sub_obj_map_h[left_or_right[0]].append(i) + else: + sub_obj_map_h[top_or_bottom[0]].append(i) + else: + if left_or_right[1] > top_or_bottom[1]: + sub_obj_map_h[top_or_bottom[0]].append(i) + else: + sub_obj_map_h[left_or_right[0]].append(i) + else: + if left_or_right[1] != float('inf'): + sub_obj_map_h[left_or_right[0]].append(i) + else: + sub_obj_map_h[top_or_bottom[0]].append(i) + ret = [] + for i in sub_obj_map_h.keys(): + ret.append( + { + 'sub_bbox': { + 'bbox': subjects[i]['bbox'], + 'score': subjects[i]['score'], + }, + 'obj_bboxes': [ + {'score': objects[j]['score'], 'bbox': objects[j]['bbox']} + for j in sub_obj_map_h[i] + ], + 'sub_idx': i, + } + ) + return ret + + def get_imgs_v2(self, page_no: int): + with_captions = self.__tie_up_category_by_distance_v2(page_no, 3, 4) + with_footnotes = self.__tie_up_category_by_distance_v2( + page_no, 3, CategoryId.ImageFootnote + ) + ret = [] + for v in with_captions: + record = { + 'image_body': v['sub_bbox'], + 'image_caption_list': v['obj_bboxes'], + } + filter_idx = v['sub_idx'] + d = next(filter(lambda x: x['sub_idx'] == filter_idx, with_footnotes)) + record['image_footnote_list'] = d['obj_bboxes'] + ret.append(record) + return ret + + def get_tables_v2(self, page_no: int) -> list: + with_captions = self.__tie_up_category_by_distance_v2(page_no, 5, 6) + with_footnotes = self.__tie_up_category_by_distance_v2(page_no, 5, 7) + ret = [] + for v in with_captions: + record = { + 'table_body': v['sub_bbox'], + 'table_caption_list': v['obj_bboxes'], + } + filter_idx = v['sub_idx'] + d = next(filter(lambda x: x['sub_idx'] == filter_idx, with_footnotes)) + record['table_footnote_list'] = d['obj_bboxes'] + ret.append(record) + return ret + def get_imgs(self, page_no: int): with_captions, _ = self.__tie_up_category_by_distance(page_no, 3, 4) with_footnotes, _ = self.__tie_up_category_by_distance( @@ -717,10 +962,10 @@ class MagicModel: def get_page_size(self, page_no: int): # 获取页面宽高 # 获取当前页的page对象 - page = self.__docs[page_no] + page = self.__docs.get_page(page_no).get_page_info() # 获取当前页的宽高 - page_w = page.rect.width - page_h = page.rect.height + page_w = page.w + page_h = page.h return page_w, page_h def __get_blocks_by_type( diff --git a/magic_pdf/model/mfr_cudagraph.py b/magic_pdf/model/mfr_cudagraph.py new file mode 100644 index 00000000..59b45c52 --- /dev/null +++ b/magic_pdf/model/mfr_cudagraph.py @@ -0,0 +1,899 @@ +from typing import Optional, Tuple, Union +import torch +from torch import nn +import os +from unimernet.common.config import Config +import unimernet.tasks as tasks +import argparse +from transformers.modeling_outputs import BaseModelOutputWithPastAndCrossAttentions +from transformers.modeling_attn_mask_utils import _prepare_4d_attention_mask, _prepare_4d_causal_attention_mask + +class PatchedMBartLearnedPositionalEmbedding(nn.Module): + + def __init__(self, origin: nn.Module): + super().__init__() + self.offset = origin.offset + self.embedding = nn.Embedding(origin.num_embeddings, origin.embedding_dim) + self.embedding.weight.data = origin.weight.data + + def forward(self, input_ids: torch.Tensor, past_key_values_length: int = 0): + """`input_ids' shape is expected to be [bsz x seqlen].""" + + bsz, seq_len = input_ids.shape[:2] + positions = torch.arange(0, seq_len, dtype=torch.long, device=self.embedding.weight.device + ) + positions += past_key_values_length + positions = positions.expand(bsz, -1) + + return self.embedding(positions + self.offset) + + +class PatchedMBartDecoder(nn.Module): + def __init__(self, origin: nn.Module, kvlen: torch.LongTensor): + super().__init__() + self.origin = origin + self.kvlen = kvlen + + self.config = origin.config + self.embed_tokens = origin.embed_tokens + self.embed_scale = origin.embed_scale + self._use_flash_attention_2 = origin._use_flash_attention_2 + self.embed_positions = origin.embed_positions + self.counting_context_weight = getattr(origin, 'counting_context_weight', None) + self.layernorm_embedding = origin.layernorm_embedding + self.layers = origin.layers + self.layer_norm = origin.layer_norm + + self.patched_embed_positions = PatchedMBartLearnedPositionalEmbedding(self.embed_positions) + + def forward( + self, + input_ids: torch.LongTensor = None, + attention_mask: Optional[torch.Tensor] = None, + count_pred: Optional[torch.FloatTensor] = None, + encoder_hidden_states: Optional[torch.FloatTensor] = None, + encoder_attention_mask: Optional[torch.LongTensor] = None, + head_mask: Optional[torch.Tensor] = None, + cross_attn_head_mask: Optional[torch.Tensor] = None, + past_key_values: Optional[Tuple[Tuple[torch.FloatTensor]]] = None, + inputs_embeds: Optional[torch.FloatTensor] = None, + use_cache: Optional[bool] = None, + output_attentions: Optional[bool] = None, + output_hidden_states: Optional[bool] = None, + return_dict: Optional[bool] = None, + ) -> Union[Tuple, BaseModelOutputWithPastAndCrossAttentions]: + run_origin = False + if past_key_values is None: + run_origin = True + elif past_key_values[0][0].size(-2) < attention_mask.size(-1): + run_origin = True + + if run_origin: + return self.origin( + input_ids=input_ids, + attention_mask=attention_mask, + count_pred=count_pred, + encoder_hidden_states=encoder_hidden_states, + encoder_attention_mask=encoder_attention_mask, + head_mask=head_mask, + cross_attn_head_mask=cross_attn_head_mask, + past_key_values=past_key_values, + inputs_embeds=inputs_embeds, + use_cache=use_cache, + output_attentions=output_attentions, + output_hidden_states=output_hidden_states, + return_dict=return_dict, + ) + + output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions + output_hidden_states = ( + output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states + ) + use_cache = use_cache if use_cache is not None else self.config.use_cache + return_dict = return_dict if return_dict is not None else self.config.use_return_dict + + # retrieve input_ids and inputs_embeds + if input_ids is not None and inputs_embeds is not None: + raise ValueError("You cannot specify both decoder_input_ids and decoder_inputs_embeds at the same time") + elif input_ids is not None: + input = input_ids + input_shape = input.size() + input_ids = input_ids.view(-1, input_shape[-1]) + elif inputs_embeds is not None: + input_shape = inputs_embeds.size()[:-1] + input = inputs_embeds[:, :, -1] + else: + raise ValueError("You have to specify either decoder_input_ids or decoder_inputs_embeds") + + # past_key_values_length + past_key_values_length = past_key_values[0][0].shape[2] if past_key_values is not None else 0 + + if inputs_embeds is None: + inputs_embeds = self.embed_tokens(input_ids) * self.embed_scale + + if self._use_flash_attention_2: + # 2d mask is passed through the layers + attention_mask = attention_mask if (attention_mask is not None and 0 in attention_mask) else None + else: + # 4d mask is passed through the layers + attention_mask = _prepare_4d_causal_attention_mask( + attention_mask, input_shape, inputs_embeds, past_key_values_length + ) + + # expand encoder attention mask + if encoder_hidden_states is not None and encoder_attention_mask is not None: + if self._use_flash_attention_2: + encoder_attention_mask = encoder_attention_mask if 0 in encoder_attention_mask else None + else: + # [bsz, seq_len] -> [bsz, 1, tgt_seq_len, src_seq_len] + encoder_attention_mask = _prepare_4d_attention_mask( + encoder_attention_mask, inputs_embeds.dtype, tgt_len=input_shape[-1] + ) + + # embed positions + positions = self.patched_embed_positions(input, self.kvlen) + + hidden_states = inputs_embeds + positions.to(inputs_embeds.device) + + # TODO: add counting context weight to hidden_states + if count_pred is not None: + count_context_weight = self.counting_context_weight(count_pred) + hidden_states = hidden_states + 0.5 * count_context_weight.unsqueeze(1) + hidden_states = self.layernorm_embedding(hidden_states) + + # decoder layers + all_hidden_states = () if output_hidden_states else None + all_self_attns = () if output_attentions else None + all_cross_attentions = () if (output_attentions and encoder_hidden_states is not None) else None + next_decoder_cache = () if use_cache else None + + # check if head_mask/cross_attn_head_mask has a correct number of layers specified if desired + for attn_mask, mask_name in zip([head_mask, cross_attn_head_mask], ["head_mask", "cross_attn_head_mask"]): + if attn_mask is not None: + if attn_mask.size()[0] != len(self.layers): + raise ValueError( + f"The `{mask_name}` should be specified for {len(self.layers)} layers, but it is for" + f" {attn_mask.size()[0]}." + ) + for idx, decoder_layer in enumerate(self.layers): + # add LayerDrop (see https://arxiv.org/abs/1909.11556 for description) + if output_hidden_states: + all_hidden_states += (hidden_states,) + + past_key_value = past_key_values[idx] if past_key_values is not None else None + layer_outputs = decoder_layer( + hidden_states, + attention_mask=attention_mask, + encoder_hidden_states=encoder_hidden_states, + encoder_attention_mask=encoder_attention_mask, + layer_head_mask=(head_mask[idx] if head_mask is not None else None), + cross_attn_layer_head_mask=( + cross_attn_head_mask[idx] if cross_attn_head_mask is not None else None + ), + past_key_value=past_key_value, + output_attentions=output_attentions, + use_cache=use_cache, + ) + hidden_states = layer_outputs[0] + + if use_cache: + next_decoder_cache += (layer_outputs[3 if output_attentions else 1],) + + if output_attentions: + all_self_attns += (layer_outputs[1],) + + if encoder_hidden_states is not None: + all_cross_attentions += (layer_outputs[2],) + + hidden_states = self.layer_norm(hidden_states) + + # add hidden states from the last decoder layer + if output_hidden_states: + all_hidden_states += (hidden_states,) + + next_cache = next_decoder_cache if use_cache else None + if not return_dict: + return tuple( + v + for v in [hidden_states, next_cache, all_hidden_states, all_self_attns, all_cross_attentions] + if v is not None + ) + return BaseModelOutputWithPastAndCrossAttentions( + last_hidden_state=hidden_states, + past_key_values=next_cache, + hidden_states=all_hidden_states, + attentions=all_self_attns, + cross_attentions=all_cross_attentions, + ) + + +class PatchedMBartAttention(nn.Module): + + def __init__(self, origin: nn.Module, kvlen: torch.LongTensor): + super().__init__() + self.embed_dim = origin.embed_dim + self.num_heads = origin.num_heads + self.dropout = origin.dropout + self.head_dim = origin.head_dim + self.config = origin.config + + self.scaling = origin.scaling + self.is_decoder = origin.is_decoder + self.is_causal = origin.is_causal + + self.k_proj = origin.k_proj + self.v_proj = origin.v_proj + self.q_proj = origin.q_proj + self.out_proj = origin.out_proj + self.kvlen = kvlen + + def _shape(self, tensor: torch.Tensor, seq_len: int, bsz: int): + return tensor.view(bsz, seq_len, self.num_heads, self.head_dim).transpose(1, 2).contiguous() + + def forward( + self, + hidden_states: torch.Tensor, + key_value_states: Optional[torch.Tensor] = None, + past_key_value: Optional[Tuple[torch.Tensor]] = None, + attention_mask: Optional[torch.Tensor] = None, + layer_head_mask: Optional[torch.Tensor] = None, + output_attentions: bool = False, + ) -> Tuple[torch.Tensor, Optional[torch.Tensor], Optional[Tuple[torch.Tensor]]]: + """Input shape: Batch x Time x Channel""" + + # if key_value_states are provided this layer is used as a cross-attention layer + # for the decoder + is_cross_attention = key_value_states is not None + + bsz, tgt_len, _ = hidden_states.size() + + # get query proj + query_states = self.q_proj(hidden_states) * self.scaling + # get key, value proj + # `past_key_value[0].shape[2] == key_value_states.shape[1]` + # is checking that the `sequence_length` of the `past_key_value` is the same as + # the provided `key_value_states` to support prefix tuning + if ( + is_cross_attention + and past_key_value is not None + and past_key_value[0].shape[2] == key_value_states.shape[1] + ): + # reuse k,v, cross_attentions + key_states = past_key_value[0] + value_states = past_key_value[1] + elif is_cross_attention: + # cross_attentions + key_states = self._shape(self.k_proj(key_value_states), -1, bsz) + value_states = self._shape(self.v_proj(key_value_states), -1, bsz) + elif past_key_value is not None: + # reuse k, v, self_attention + key_states = self._shape(self.k_proj(hidden_states), -1, bsz) + value_states = self._shape(self.v_proj(hidden_states), -1, bsz) + + if past_key_value[0].size(-2) < attention_mask.size(-1): + key_states = torch.cat([past_key_value[0], key_states], dim=2) + value_states = torch.cat([past_key_value[1], value_states], dim=2) + else: + past_key_value[0][:, :, self.kvlen[None]] = key_states + past_key_value[1][:, :, self.kvlen[None]] = value_states + key_states = past_key_value[0] + value_states = past_key_value[1] + else: + # self_attention + key_states = self._shape(self.k_proj(hidden_states), -1, bsz) + value_states = self._shape(self.v_proj(hidden_states), -1, bsz) + + if self.is_decoder: + past_key_value = (key_states, value_states) + + proj_shape = (bsz * self.num_heads, -1, self.head_dim) + query_states = self._shape(query_states, tgt_len, bsz).view(*proj_shape) + key_states = key_states.reshape(*proj_shape) + value_states = value_states.reshape(*proj_shape) + + src_len = key_states.size(1) + attn_weights = torch.bmm(query_states, key_states.transpose(1, 2)) + + if attn_weights.size() != (bsz * self.num_heads, tgt_len, src_len): + raise ValueError( + f"Attention weights should be of size {(bsz * self.num_heads, tgt_len, src_len)}, but is" + f" {attn_weights.size()}" + ) + + if attention_mask is not None: + if attention_mask.size() != (bsz, 1, tgt_len, src_len): + raise ValueError( + f"Attention mask should be of size {(bsz, 1, tgt_len, src_len)}, but is {attention_mask.size()}" + ) + attn_weights = attn_weights.view(bsz, self.num_heads, tgt_len, src_len) + attention_mask + attn_weights = attn_weights.view(bsz * self.num_heads, tgt_len, src_len) + + attn_weights = nn.functional.softmax(attn_weights, dim=-1) + + if layer_head_mask is not None: + if layer_head_mask.size() != (self.num_heads,): + raise ValueError( + f"Head mask for a single layer should be of size {(self.num_heads,)}, but is" + f" {layer_head_mask.size()}" + ) + attn_weights = layer_head_mask.view(1, -1, 1, 1) * attn_weights.view(bsz, self.num_heads, tgt_len, src_len) + attn_weights = attn_weights.view(bsz * self.num_heads, tgt_len, src_len) + + if output_attentions: + # this operation is a bit awkward, but it's required to + # make sure that attn_weights keeps its gradient. + # In order to do so, attn_weights have to be reshaped + # twice and have to be reused in the following + attn_weights_reshaped = attn_weights.view(bsz, self.num_heads, tgt_len, src_len) + attn_weights = attn_weights_reshaped.view(bsz * self.num_heads, tgt_len, src_len) + else: + attn_weights_reshaped = None + + attn_probs = attn_weights + + attn_output = torch.bmm(attn_probs, value_states) + + if attn_output.size() != (bsz * self.num_heads, tgt_len, self.head_dim): + raise ValueError( + f"`attn_output` should be of size {(bsz * self.num_heads, tgt_len, self.head_dim)}, but is" + f" {attn_output.size()}" + ) + + attn_output = attn_output.view(bsz, self.num_heads, tgt_len, self.head_dim) + attn_output = attn_output.transpose(1, 2) + + # Use the `embed_dim` from the config (stored in the class) rather than `hidden_state` because `attn_output` can be + # partitioned across GPUs when using tensor-parallelism. + attn_output = attn_output.reshape(bsz, tgt_len, self.embed_dim) + + # attn_output = self.out_proj(attn_output) + attn_output = self.out_proj(attn_output) + + return attn_output, attn_weights_reshaped, past_key_value + + +class PatchedMBartSqueezeAttention(nn.Module): + + def __init__(self, origin: nn.Module, kvlen: torch.LongTensor): + super().__init__() + self.embed_dim = origin.embed_dim + self.num_heads = origin.num_heads + self.dropout = origin.dropout + self.head_dim = origin.head_dim + self.squeeze_head_dim=origin.squeeze_head_dim + self.config = origin.config + + self.scaling = origin.scaling + self.is_decoder = origin.is_decoder + self.scaling = origin.scaling + + self.q_proj = origin.q_proj + self.k_proj = origin.k_proj + self.v_proj = origin.v_proj + self.out_proj = origin.out_proj + self.kvlen = kvlen + + def _shape_qk(self, tensor: torch.Tensor, seq_len: int, bsz: int): + return tensor.view(bsz, seq_len, self.num_heads, self.squeeze_head_dim).transpose(1, 2).contiguous() + + def _shape_v(self, tensor: torch.Tensor, seq_len: int, bsz: int): + return tensor.view(bsz, seq_len, self.num_heads, self.head_dim).transpose(1, 2).contiguous() + + def forward( + self, + hidden_states: torch.Tensor, + key_value_states: Optional[torch.Tensor] = None, + past_key_value: Optional[Tuple[torch.Tensor]] = None, + attention_mask: Optional[torch.Tensor] = None, + layer_head_mask: Optional[torch.Tensor] = None, + output_attentions: bool = False, + ) -> Tuple[torch.Tensor, Optional[torch.Tensor], Optional[Tuple[torch.Tensor]]]: + """Input shape: Batch x Time x Channel""" + + # if key_value_states are provided this layer is used as a cross-attention layer + # for the decoder + is_cross_attention = key_value_states is not None + + bsz, tgt_len, _ = hidden_states.size() + + # get query proj + query_states = self.q_proj(hidden_states) * self.scaling + # get key, value proj + # `past_key_value[0].shape[2] == key_value_states.shape[1]` + # is checking that the `sequence_length` of the `past_key_value` is the same as + # the provided `key_value_states` to support prefix tuning + if ( + is_cross_attention + and past_key_value is not None + and past_key_value[0].shape[2] == key_value_states.shape[1] + ): + # reuse k,v, cross_attentions + key_states = past_key_value[0] + value_states = past_key_value[1] + elif is_cross_attention: + # cross_attentions + key_states = self._shape_qk(self.k_proj(key_value_states), -1, bsz) + value_states = self._shape_v(self.v_proj(key_value_states), -1, bsz) + elif past_key_value is not None: + # reuse k, v, self_attention + key_states = self._shape_qk(self.k_proj(hidden_states), -1, bsz) + value_states = self._shape_v(self.v_proj(hidden_states), -1, bsz) + + if past_key_value[0].size(-2) < attention_mask.size(-1): + key_states = torch.cat([past_key_value[0], key_states], dim=2) + value_states = torch.cat([past_key_value[1], value_states], dim=2) + else: + past_key_value[0][:, :, self.kvlen[None]] = key_states + past_key_value[1][:, :, self.kvlen[None]] = value_states + key_states = past_key_value[0] + value_states = past_key_value[1] + else: + # self_attention + key_states = self._shape_qk(self.k_proj(hidden_states), -1, bsz) + value_states = self._shape_v(self.v_proj(hidden_states), -1, bsz) + + if self.is_decoder: + # if cross_attention save Tuple(torch.Tensor, torch.Tensor) of all cross attention key/value_states. + # Further calls to cross_attention layer can then reuse all cross-attention + # key/value_states (first "if" case) + # if uni-directional self-attention (decoder) save Tuple(torch.Tensor, torch.Tensor) of + # all previous decoder key/value_states. Further calls to uni-directional self-attention + # can concat previous decoder key/value_states to current projected key/value_states (third "elif" case) + # if encoder bi-directional self-attention `past_key_value` is always `None` + past_key_value = (key_states, value_states) + + proj_shape = (bsz * self.num_heads, -1, self.squeeze_head_dim) + value_shape = (bsz * self.num_heads, -1, self.head_dim) + query_states = self._shape_qk(query_states, tgt_len, bsz).view(*proj_shape) + key_states = key_states.reshape(*proj_shape) + value_states = value_states.reshape(*value_shape) + + src_len = key_states.size(1) + attn_weights = torch.bmm(query_states, key_states.transpose(1, 2)) + + if attn_weights.size() != (bsz * self.num_heads, tgt_len, src_len): + raise ValueError( + f"Attention weights should be of size {(bsz * self.num_heads, tgt_len, src_len)}, but is" + f" {attn_weights.size()}" + ) + + if attention_mask is not None: + if attention_mask.size() != (bsz, 1, tgt_len, src_len): + raise ValueError( + f"Attention mask should be of size {(bsz, 1, tgt_len, src_len)}, but is {attention_mask.size()}" + ) + attn_weights = attn_weights.view(bsz, self.num_heads, tgt_len, src_len) + attention_mask + attn_weights = attn_weights.view(bsz * self.num_heads, tgt_len, src_len) + + attn_weights = nn.functional.softmax(attn_weights, dim=-1) + + if layer_head_mask is not None: + if layer_head_mask.size() != (self.num_heads,): + raise ValueError( + f"Head mask for a single layer should be of size {(self.num_heads,)}, but is" + f" {layer_head_mask.size()}" + ) + attn_weights = layer_head_mask.view(1, -1, 1, 1) * attn_weights.view(bsz, self.num_heads, tgt_len, src_len) + attn_weights = attn_weights.view(bsz * self.num_heads, tgt_len, src_len) + + if output_attentions: + # this operation is a bit awkward, but it's required to + # make sure that attn_weights keeps its gradient. + # In order to do so, attn_weights have to be reshaped + # twice and have to be reused in the following + attn_weights_reshaped = attn_weights.view(bsz, self.num_heads, tgt_len, src_len) + attn_weights = attn_weights_reshaped.view(bsz * self.num_heads, tgt_len, src_len) + else: + attn_weights_reshaped = None + + attn_probs = nn.functional.dropout(attn_weights, p=self.dropout, training=self.training) + + attn_output = torch.bmm(attn_probs, value_states) + + if attn_output.size() != (bsz * self.num_heads, tgt_len, self.head_dim): + raise ValueError( + f"`attn_output` should be of size {(bsz * self.num_heads, tgt_len, self.head_dim)}, but is" + f" {attn_output.size()}" + ) + + attn_output = attn_output.view(bsz, self.num_heads, tgt_len, self.head_dim) + attn_output = attn_output.transpose(1, 2) + + # Use the `embed_dim` from the config (stored in the class) rather than `hidden_state` because `attn_output` can be + # partitioned across GPUs when using tensor-parallelism. + attn_output = attn_output.reshape(bsz, tgt_len, self.embed_dim) + + attn_output = self.out_proj(attn_output) + + return attn_output, attn_weights_reshaped, past_key_value + +def patch_model(model: nn.Module, kvlen: torch.LongTensor): + for name, child in model.named_children(): + cls_name = type(child).__name__ + if cls_name == 'MBartAttention': + patched_child = PatchedMBartAttention(child, kvlen) + model.register_module(name, patched_child) + elif cls_name == 'MBartSqueezeAttention': + patched_child = PatchedMBartSqueezeAttention(child, kvlen) + model.register_module(name, patched_child) + else: + patch_model(child, kvlen) + + cls_name = type(model).__name__ + if cls_name == 'CustomMBartDecoder': + model = PatchedMBartDecoder(model, kvlen) + return model + + +def next_power_of_2(n: int): + """Return the smallest power of 2 greater than or equal to n.""" + n -= 1 + n |= n >> 1 + n |= n >> 2 + n |= n >> 4 + n |= n >> 8 + n |= n >> 16 + n |= n >> 32 + n += 1 + return n + + +def get_graph_key(batch_size: int, kvlens: int): + batch_size = next_power_of_2(batch_size) + kvlens = next_power_of_2(kvlens) + + batch_size = max(8, batch_size) + kvlens = max(32, kvlens) + + return batch_size, kvlens + + +class GraphRunnerImpl: + + def __init__(self, model: nn.Module, graph: torch.cuda.CUDAGraph, input_buffers: dict, output_buffers: dict): + self.model = model + self.graph = graph + self.input_buffers = input_buffers + self.output_buffers = output_buffers + + @staticmethod + def extract_input_buffers(input_buffers: dict, batch_size: int, kvlens: int): + input_ids = input_buffers['input_ids'][:batch_size] + attention_mask = input_buffers['attention_mask'][:batch_size, :kvlens] + encoder_hidden_states = input_buffers['encoder_hidden_states'][:batch_size] + kvlen=input_buffers['kvlen'] + + past_key_values = [] + for past_key_value in input_buffers['past_key_values']: + k0 = past_key_value[0][:batch_size, :, :kvlens] + v0 = past_key_value[1][:batch_size, :, :kvlens] + k1 = past_key_value[2][:batch_size] + v1 = past_key_value[3][:batch_size] + past_key_values.append((k0, v0, k1, v1)) + + input_buffers = dict( + input_ids=input_ids, + attention_mask=attention_mask, + encoder_hidden_states=encoder_hidden_states, + past_key_values=past_key_values, + kvlen=kvlen, + ) + return input_buffers + + @staticmethod + def fill_input_buffers( + input_buffer: dict, + input_ids: torch.LongTensor = None, + attention_mask: Optional[torch.Tensor] = None, + encoder_hidden_states: Optional[torch.FloatTensor] = None, + past_key_values: Optional[Tuple[Tuple[torch.FloatTensor]]] = None, + ): + batch_size = input_ids.size(0) + kvlens = attention_mask.size(1) + + input_buffer['input_ids'][:batch_size] = input_ids + + if input_buffer['attention_mask'].data_ptr() != attention_mask.data_ptr(): + input_buffer['attention_mask'].fill_(0) + input_buffer['attention_mask'][:batch_size, :kvlens] = attention_mask + input_buffer['encoder_hidden_states'][:batch_size] = encoder_hidden_states + + if past_key_values is not None: + for buf_kv, kv in zip(input_buffer['past_key_values'], past_key_values): + idx = 0 + if buf_kv[idx].data_ptr() != kv[idx].data_ptr(): + buf_kv[idx].fill_(0) + buf_kv[idx][:batch_size, :, :kvlens-1] = kv[idx] + idx = 1 + if buf_kv[idx].data_ptr() != kv[idx].data_ptr(): + buf_kv[idx].fill_(0) + buf_kv[idx][:batch_size, :, :kvlens-1] = kv[idx] + + idx = 2 + if buf_kv[idx].data_ptr() != kv[idx].data_ptr(): + buf_kv[idx].fill_(0) + buf_kv[idx][:batch_size] = kv[idx] + idx = 3 + if buf_kv[idx].data_ptr() != kv[idx].data_ptr(): + buf_kv[idx].fill_(0) + buf_kv[idx][:batch_size] = kv[idx] + + input_buffer['kvlen'].fill_(kvlens - 1) + + @classmethod + @torch.inference_mode() + def capture(cls, + model: nn.Module, + input_buffers: dict, + pool, + warmup: bool = False, + input_ids: torch.LongTensor = None, + attention_mask: Optional[torch.Tensor] = None, + count_pred: Optional[torch.FloatTensor] = None, + encoder_hidden_states: Optional[torch.FloatTensor] = None, + encoder_attention_mask: Optional[torch.LongTensor] = None, + head_mask: Optional[torch.Tensor] = None, + cross_attn_head_mask: Optional[torch.Tensor] = None, + past_key_values: Optional[Tuple[Tuple[torch.FloatTensor]]] = None, + inputs_embeds: Optional[torch.FloatTensor] = None, + use_cache: Optional[bool] = None, + output_attentions: Optional[bool] = None, + output_hidden_states: Optional[bool] = None, + return_dict: Optional[bool] = None,): + batch_size = input_ids.size(0) + kvlens = attention_mask.size(1) + + graph_key = get_graph_key(batch_size, kvlens) + batch_size = graph_key[0] + kvlens = graph_key[1] + + input_buffers = cls.extract_input_buffers(input_buffers, + batch_size=batch_size, + kvlens=kvlens) + cls.fill_input_buffers(input_buffers, + input_ids, + attention_mask, + encoder_hidden_states, + past_key_values) + + input_ids = input_buffers['input_ids'] + attention_mask = input_buffers['attention_mask'] + encoder_hidden_states = input_buffers['encoder_hidden_states'] + past_key_values = input_buffers['past_key_values'] + + if warmup: + # warmup + model( + input_ids=input_ids, + attention_mask=attention_mask, + count_pred=count_pred, + encoder_hidden_states=encoder_hidden_states, + encoder_attention_mask=encoder_attention_mask, + head_mask=head_mask, + cross_attn_head_mask=cross_attn_head_mask, + past_key_values=past_key_values, + inputs_embeds=inputs_embeds, + use_cache=use_cache, + output_attentions=output_attentions, + output_hidden_states=output_hidden_states, + return_dict=return_dict) + + graph = torch.cuda.CUDAGraph() + with torch.cuda.graph(graph, + pool=pool): + outputs = model( + input_ids=input_ids, + attention_mask=attention_mask, + count_pred=count_pred, + encoder_hidden_states=encoder_hidden_states, + encoder_attention_mask=encoder_attention_mask, + head_mask=head_mask, + cross_attn_head_mask=cross_attn_head_mask, + past_key_values=past_key_values, + inputs_embeds=inputs_embeds, + use_cache=use_cache, + output_attentions=output_attentions, + output_hidden_states=output_hidden_states, + return_dict=return_dict) + + output_buffers = dict( + last_hidden_state=outputs['last_hidden_state'], + past_key_values=outputs['past_key_values'], + ) + + return GraphRunnerImpl(model, graph, input_buffers, output_buffers) + + def __call__(self, + input_ids: torch.LongTensor = None, + attention_mask: Optional[torch.Tensor] = None, + count_pred: Optional[torch.FloatTensor] = None, + encoder_hidden_states: Optional[torch.FloatTensor] = None, + encoder_attention_mask: Optional[torch.LongTensor] = None, + head_mask: Optional[torch.Tensor] = None, + cross_attn_head_mask: Optional[torch.Tensor] = None, + past_key_values: Optional[Tuple[Tuple[torch.FloatTensor]]] = None, + inputs_embeds: Optional[torch.FloatTensor] = None, + use_cache: Optional[bool] = None, + output_attentions: Optional[bool] = None, + output_hidden_states: Optional[bool] = None, + return_dict: Optional[bool] = None, + ): + batch_size = input_ids.size(0) + kvlens = attention_mask.size(1) + self.fill_input_buffers(self.input_buffers, + input_ids, + attention_mask, + encoder_hidden_states, + past_key_values) + + self.graph.replay() + + last_hidden_state = self.output_buffers['last_hidden_state'][:batch_size] + + past_key_values = [] + for past_key_value in self.output_buffers['past_key_values']: + k0 = past_key_value[0][:batch_size, :, :kvlens] + v0 = past_key_value[1][:batch_size, :, :kvlens] + k1 = past_key_value[2][:batch_size] + v1 = past_key_value[3][:batch_size] + past_key_values.append((k0, v0, k1, v1)) + + outputs = BaseModelOutputWithPastAndCrossAttentions( + last_hidden_state=last_hidden_state, + past_key_values=past_key_values, + ) + return outputs + +class GraphRunner(nn.Module): + + def __init__(self, model: nn.Module, max_batchs: int, max_kvlens: int, dtype:torch.dtype = torch.float16, device: torch.device = 'cuda'): + super().__init__() + + self.kvlen = torch.tensor(0, dtype=torch.long, device=device) + model = patch_model(model.to(dtype), self.kvlen) + self.model = model + self.max_batchs = max_batchs + self.max_kvlens = max_kvlens + self.device = device + + self.input_buffers = None + + self.impl_map = dict() + self.graph_pool_handle = torch.cuda.graph_pool_handle() + self.warmuped = False + + def create_buffers(self, encoder_kvlens: int, dtype: torch.dtype): + max_batchs = self.max_batchs + max_kvlens = self.max_kvlens + device = self.device + config = self.model.config + + d_model = config.d_model + decoder_layers = config.decoder_layers + num_heads = config.decoder_attention_heads + + head_dim = d_model // num_heads + self_attn = self.model.layers[0].self_attn + qk_head_dim = getattr(self_attn, 'squeeze_head_dim', head_dim) + + input_ids = torch.ones((max_batchs, 1), dtype=torch.int64, device=device) + attention_mask = torch.zeros((max_batchs, max_kvlens), dtype=torch.int64, device=device) + encoder_hidden_states = torch.zeros((max_batchs, encoder_kvlens, d_model), dtype=dtype, device=device) + + past_key_values = [] + for _ in range(decoder_layers): + k0 = torch.zeros((max_batchs, num_heads, max_kvlens, qk_head_dim), dtype=dtype, device=device) + v0 = torch.zeros((max_batchs, num_heads, max_kvlens, head_dim), dtype=dtype, device=device) + k1 = torch.zeros((max_batchs, num_heads, encoder_kvlens, qk_head_dim), dtype=dtype, device=device) + v1 = torch.zeros((max_batchs, num_heads, encoder_kvlens, head_dim), dtype=dtype, device=device) + + past_key_values.append((k0, v0, k1, v1)) + + self.input_buffers = dict( + input_ids=input_ids, + attention_mask=attention_mask, + encoder_hidden_states=encoder_hidden_states, + past_key_values=past_key_values, + kvlen=self.kvlen + ) + + @torch.inference_mode() + def forward(self, + input_ids: torch.LongTensor = None, + attention_mask: Optional[torch.Tensor] = None, + count_pred: Optional[torch.FloatTensor] = None, + encoder_hidden_states: Optional[torch.FloatTensor] = None, + encoder_attention_mask: Optional[torch.LongTensor] = None, + head_mask: Optional[torch.Tensor] = None, + cross_attn_head_mask: Optional[torch.Tensor] = None, + past_key_values: Optional[Tuple[Tuple[torch.FloatTensor]]] = None, + inputs_embeds: Optional[torch.FloatTensor] = None, + use_cache: Optional[bool] = None, + output_attentions: Optional[bool] = None, + output_hidden_states: Optional[bool] = None, + return_dict: Optional[bool] = None, + ): + batch_size, qlens = input_ids.size() + kvlens = attention_mask.size(1) + + eager_mode = False + + if qlens != 1: + eager_mode = True + + if past_key_values is None: + eager_mode = True + else: + for past_key_value in past_key_values: + if past_key_value is None: + eager_mode = True + break + + if batch_size >= self.max_batchs or kvlens >= self.max_kvlens: + eager_mode = True + + if eager_mode: + return self.model( + input_ids=input_ids, + attention_mask=attention_mask, + count_pred=count_pred, + encoder_hidden_states=encoder_hidden_states, + encoder_attention_mask=encoder_attention_mask, + head_mask=head_mask, + cross_attn_head_mask=cross_attn_head_mask, + past_key_values=past_key_values, + inputs_embeds=inputs_embeds, + use_cache=use_cache, + output_attentions=output_attentions, + output_hidden_states=output_hidden_states, + return_dict=return_dict,) + + # create buffer if not exists. + if self.input_buffers is None: + encoder_kvlens = encoder_hidden_states.size(1) + self.create_buffers(encoder_kvlens=encoder_kvlens, dtype=encoder_hidden_states.dtype) + + graph_key = get_graph_key(batch_size, kvlens) + if graph_key not in self.impl_map: + warmup = False + if not self.warmuped: + warmup = True + self.warmuped = True + impl = GraphRunnerImpl.capture( + self.model, + self.input_buffers, + self.graph_pool_handle, + warmup=warmup, + input_ids=input_ids, + attention_mask=attention_mask, + count_pred=count_pred, + encoder_hidden_states=encoder_hidden_states, + encoder_attention_mask=encoder_attention_mask, + head_mask=head_mask, + cross_attn_head_mask=cross_attn_head_mask, + past_key_values=past_key_values, + inputs_embeds=inputs_embeds, + use_cache=use_cache, + output_attentions=output_attentions, + output_hidden_states=output_hidden_states, + return_dict=return_dict, + ) + self.impl_map[graph_key] = impl + impl = self.impl_map[graph_key] + + ret = impl( + input_ids=input_ids, + attention_mask=attention_mask, + count_pred=count_pred, + encoder_hidden_states=encoder_hidden_states, + encoder_attention_mask=encoder_attention_mask, + head_mask=head_mask, + cross_attn_head_mask=cross_attn_head_mask, + past_key_values=past_key_values, + inputs_embeds=inputs_embeds, + use_cache=use_cache, + output_attentions=output_attentions, + output_hidden_states=output_hidden_states, + return_dict=return_dict, + ) + return ret \ No newline at end of file diff --git a/magic_pdf/model/pdf_extract_kit.py b/magic_pdf/model/pdf_extract_kit.py index 0c296fba..fb3a5f79 100644 --- a/magic_pdf/model/pdf_extract_kit.py +++ b/magic_pdf/model/pdf_extract_kit.py @@ -6,6 +6,7 @@ import shutil from magic_pdf.libs.Constants import * from magic_pdf.libs.clean_memory import clean_memory from magic_pdf.model.model_list import AtomicModel +from .mfr_cudagraph import GraphRunner os.environ['NO_ALBUMENTATIONS_UPDATE'] = '1' # 禁止albumentations检查更新 os.environ['YOLO_VERBOSE'] = 'False' # disable yolo logger @@ -26,6 +27,7 @@ try: from unimernet.common.config import Config import unimernet.tasks as tasks from unimernet.processors import load_processor + from doclayout_yolo import YOLOv10 except ImportError as e: logger.exception(e) @@ -42,7 +44,7 @@ from magic_pdf.model.ppTableModel import ppTableModel def table_model_init(table_model_type, model_path, max_time, _device_='cpu'): - if table_model_type == STRUCT_EQTABLE: + if table_model_type == MODEL_NAME.STRUCT_EQTABLE: table_model = StructTableModel(model_path, max_time=max_time, device=_device_) else: config = { @@ -68,6 +70,11 @@ def mfr_model_init(weight_dir, cfg_path, _device_='cpu'): model = task.build_model(cfg) model.to(_device_) model.eval() + model = model.to(_device_) + if 'cuda' in _device_: + decoder_runner = GraphRunner(model.model.model.decoder.model.decoder, max_batchs=128, max_kvlens=256, + device=_device_) + model.model.model.decoder.model.decoder = decoder_runner vis_processor = load_processor('formula_image_eval', cfg.config.datasets.formula_rec_eval.vis_processor.eval) mfr_transform = transforms.Compose([vis_processor, ]) return [model, mfr_transform] @@ -78,11 +85,16 @@ def layout_model_init(weight, config_file, device): return model -def ocr_model_init(show_log: bool = False, det_db_box_thresh=0.3, lang=None): +def doclayout_yolo_model_init(weight): + model = YOLOv10(weight) + return model + + +def ocr_model_init(show_log: bool = False, det_db_box_thresh=0.3, lang=None, use_dilation=True, det_db_unclip_ratio=1.8): if lang is not None: - model = ModifiedPaddleOCR(show_log=show_log, det_db_box_thresh=det_db_box_thresh, lang=lang) + model = ModifiedPaddleOCR(show_log=show_log, det_db_box_thresh=det_db_box_thresh, lang=lang, use_dilation=use_dilation, det_db_unclip_ratio=det_db_unclip_ratio) else: - model = ModifiedPaddleOCR(show_log=show_log, det_db_box_thresh=det_db_box_thresh) + model = ModifiedPaddleOCR(show_log=show_log, det_db_box_thresh=det_db_box_thresh, use_dilation=use_dilation, det_db_unclip_ratio=det_db_unclip_ratio) return model @@ -115,19 +127,27 @@ class AtomModelSingleton: return cls._instance def get_atom_model(self, atom_model_name: str, **kwargs): - if atom_model_name not in self._models: - self._models[atom_model_name] = atom_model_init(model_name=atom_model_name, **kwargs) - return self._models[atom_model_name] + lang = kwargs.get("lang", None) + layout_model_name = kwargs.get("layout_model_name", None) + key = (atom_model_name, layout_model_name, lang) + if key not in self._models: + self._models[key] = atom_model_init(model_name=atom_model_name, **kwargs) + return self._models[key] def atom_model_init(model_name: str, **kwargs): if model_name == AtomicModel.Layout: - atom_model = layout_model_init( - kwargs.get("layout_weights"), - kwargs.get("layout_config_file"), - kwargs.get("device") - ) + if kwargs.get("layout_model_name") == MODEL_NAME.LAYOUTLMv3: + atom_model = layout_model_init( + kwargs.get("layout_weights"), + kwargs.get("layout_config_file"), + kwargs.get("device") + ) + elif kwargs.get("layout_model_name") == MODEL_NAME.DocLayout_YOLO: + atom_model = doclayout_yolo_model_init( + kwargs.get("doclayout_yolo_weights"), + ) elif model_name == AtomicModel.MFD: atom_model = mfd_model_init( kwargs.get("mfd_weights") @@ -146,7 +166,7 @@ def atom_model_init(model_name: str, **kwargs): ) elif model_name == AtomicModel.Table: atom_model = table_model_init( - kwargs.get("table_model_type"), + kwargs.get("table_model_name"), kwargs.get("table_model_path"), kwargs.get("table_max_time"), kwargs.get("device") @@ -194,23 +214,35 @@ class CustomPEKModel: with open(config_path, "r", encoding='utf-8') as f: self.configs = yaml.load(f, Loader=yaml.FullLoader) # 初始化解析配置 - self.apply_layout = kwargs.get("apply_layout", self.configs["config"]["layout"]) - self.apply_formula = kwargs.get("apply_formula", self.configs["config"]["formula"]) + + # layout config + self.layout_config = kwargs.get("layout_config") + self.layout_model_name = self.layout_config.get("model", MODEL_NAME.DocLayout_YOLO) + + # formula config + self.formula_config = kwargs.get("formula_config") + self.mfd_model_name = self.formula_config.get("mfd_model", MODEL_NAME.YOLO_V8_MFD) + self.mfr_model_name = self.formula_config.get("mfr_model", MODEL_NAME.UniMerNet_v2_Small) + self.apply_formula = self.formula_config.get("enable", True) + # table config - self.table_config = kwargs.get("table_config", self.configs["config"]["table_config"]) - self.apply_table = self.table_config.get("is_table_recog_enable", False) + self.table_config = kwargs.get("table_config") + self.apply_table = self.table_config.get("enable", False) self.table_max_time = self.table_config.get("max_time", TABLE_MAX_TIME_VALUE) - self.table_model_type = self.table_config.get("model", TABLE_MASTER) + self.table_model_name = self.table_config.get("model", MODEL_NAME.TABLE_MASTER) + + # ocr config self.apply_ocr = ocr self.lang = kwargs.get("lang", None) + logger.info( - "DocAnalysis init, this may take some times. apply_layout: {}, apply_formula: {}, apply_ocr: {}, apply_table: {}, lang: {}".format( - self.apply_layout, self.apply_formula, self.apply_ocr, self.apply_table, self.lang + "DocAnalysis init, this may take some times, layout_model: {}, apply_formula: {}, apply_ocr: {}, " + "apply_table: {}, table_model: {}, lang: {}".format( + self.layout_model_name, self.apply_formula, self.apply_ocr, self.apply_table, self.table_model_name, self.lang ) ) - assert self.apply_layout, "DocAnalysis must contain layout model." # 初始化解析方案 - self.device = kwargs.get("device", self.configs["config"]["device"]) + self.device = kwargs.get("device", "cpu") logger.info("using device: {}".format(self.device)) models_dir = kwargs.get("models_dir", os.path.join(root_dir, "resources", "models")) logger.info("using models_dir: {}".format(models_dir)) @@ -219,17 +251,16 @@ class CustomPEKModel: # 初始化公式识别 if self.apply_formula: + # 初始化公式检测模型 - # self.mfd_model = mfd_model_init(str(os.path.join(models_dir, self.configs["weights"]["mfd"]))) self.mfd_model = atom_model_manager.get_atom_model( atom_model_name=AtomicModel.MFD, - mfd_weights=str(os.path.join(models_dir, self.configs["weights"]["mfd"])) + mfd_weights=str(os.path.join(models_dir, self.configs["weights"][self.mfd_model_name])) ) + # 初始化公式解析模型 - mfr_weight_dir = str(os.path.join(models_dir, self.configs["weights"]["mfr"])) + mfr_weight_dir = str(os.path.join(models_dir, self.configs["weights"][self.mfr_model_name])) mfr_cfg_path = str(os.path.join(model_config_dir, "UniMERNet", "demo.yaml")) - # self.mfr_model, mfr_vis_processors = mfr_model_init(mfr_weight_dir, mfr_cfg_path, _device_=self.device) - # self.mfr_transform = transforms.Compose([mfr_vis_processors, ]) self.mfr_model, self.mfr_transform = atom_model_manager.get_atom_model( atom_model_name=AtomicModel.MFR, mfr_weight_dir=mfr_weight_dir, @@ -238,17 +269,20 @@ class CustomPEKModel: ) # 初始化layout模型 - # self.layout_model = Layoutlmv3_Predictor( - # str(os.path.join(models_dir, self.configs['weights']['layout'])), - # str(os.path.join(model_config_dir, "layoutlmv3", "layoutlmv3_base_inference.yaml")), - # device=self.device - # ) - self.layout_model = atom_model_manager.get_atom_model( - atom_model_name=AtomicModel.Layout, - layout_weights=str(os.path.join(models_dir, self.configs['weights']['layout'])), - layout_config_file=str(os.path.join(model_config_dir, "layoutlmv3", "layoutlmv3_base_inference.yaml")), - device=self.device - ) + if self.layout_model_name == MODEL_NAME.LAYOUTLMv3: + self.layout_model = atom_model_manager.get_atom_model( + atom_model_name=AtomicModel.Layout, + layout_model_name=MODEL_NAME.LAYOUTLMv3, + layout_weights=str(os.path.join(models_dir, self.configs['weights'][self.layout_model_name])), + layout_config_file=str(os.path.join(model_config_dir, "layoutlmv3", "layoutlmv3_base_inference.yaml")), + device=self.device + ) + elif self.layout_model_name == MODEL_NAME.DocLayout_YOLO: + self.layout_model = atom_model_manager.get_atom_model( + atom_model_name=AtomicModel.Layout, + layout_model_name=MODEL_NAME.DocLayout_YOLO, + doclayout_yolo_weights=str(os.path.join(models_dir, self.configs['weights'][self.layout_model_name])) + ) # 初始化ocr if self.apply_ocr: @@ -261,12 +295,10 @@ class CustomPEKModel: ) # init table model if self.apply_table: - table_model_dir = self.configs["weights"][self.table_model_type] - # self.table_model = table_model_init(self.table_model_type, str(os.path.join(models_dir, table_model_dir)), - # max_time=self.table_max_time, _device_=self.device) + table_model_dir = self.configs["weights"][self.table_model_name] self.table_model = atom_model_manager.get_atom_model( atom_model_name=AtomicModel.Table, - table_model_type=self.table_model_type, + table_model_name=self.table_model_name, table_model_path=str(os.path.join(models_dir, table_model_dir)), table_max_time=self.table_max_time, device=self.device @@ -294,7 +326,21 @@ class CustomPEKModel: # layout检测 layout_start = time.time() - layout_res = self.layout_model(image, ignore_catids=[]) + if self.layout_model_name == MODEL_NAME.LAYOUTLMv3: + # layoutlmv3 + layout_res = self.layout_model(image, ignore_catids=[]) + elif self.layout_model_name == MODEL_NAME.DocLayout_YOLO: + # doclayout_yolo + layout_res = [] + doclayout_yolo_res = self.layout_model.predict(image, imgsz=1024, conf=0.25, iou=0.45, verbose=True, device=self.device)[0] + for xyxy, conf, cla in zip(doclayout_yolo_res.boxes.xyxy.cpu(), doclayout_yolo_res.boxes.conf.cpu(), doclayout_yolo_res.boxes.cls.cpu()): + xmin, ymin, xmax, ymax = [int(p.item()) for p in xyxy] + new_item = { + 'category_id': int(cla.item()), + 'poly': [xmin, ymin, xmax, ymin, xmax, ymax, xmin, ymax], + 'score': round(float(conf.item()), 3), + } + layout_res.append(new_item) layout_cost = round(time.time() - layout_start, 2) logger.info(f"layout detection time: {layout_cost}") @@ -303,7 +349,7 @@ class CustomPEKModel: if self.apply_formula: # 公式检测 mfd_start = time.time() - mfd_res = self.mfd_model.predict(image, imgsz=1888, conf=0.25, iou=0.45, verbose=True)[0] + mfd_res = self.mfd_model.predict(image, imgsz=1888, conf=0.25, iou=0.45, verbose=True, device=self.device)[0] logger.info(f"mfd time: {round(time.time() - mfd_start, 2)}") for xyxy, conf, cla in zip(mfd_res.boxes.xyxy.cpu(), mfd_res.boxes.conf.cpu(), mfd_res.boxes.cls.cpu()): xmin, ymin, xmax, ymax = [int(p.item()) for p in xyxy] @@ -315,7 +361,6 @@ class CustomPEKModel: } layout_res.append(new_item) latex_filling_list.append(new_item) - # bbox_img = get_croped_image(pil_img, [xmin, ymin, xmax, ymax]) bbox_img = pil_img.crop((xmin, ymin, xmax, ymax)) mf_image_list.append(bbox_img) @@ -417,7 +462,7 @@ class CustomPEKModel: # logger.info("------------------table recognition processing begins-----------------") latex_code = None html_code = None - if self.table_model_type == STRUCT_EQTABLE: + if self.table_model_name == MODEL_NAME.STRUCT_EQTABLE: with torch.no_grad(): latex_code = self.table_model.image2latex(new_image)[0] else: diff --git a/magic_pdf/model/ppTableModel.py b/magic_pdf/model/ppTableModel.py index 310bcc79..933f31a0 100644 --- a/magic_pdf/model/ppTableModel.py +++ b/magic_pdf/model/ppTableModel.py @@ -52,11 +52,11 @@ class ppTableModel(object): rec_model_dir = os.path.join(model_dir, REC_MODEL_DIR) rec_char_dict_path = os.path.join(model_dir, REC_CHAR_DICT) device = kwargs.get("device", "cpu") - use_gpu = True if device == "cuda" else False + use_gpu = True if device.startswith("cuda") else False config = { "use_gpu": use_gpu, "table_max_len": kwargs.get("table_max_len", TABLE_MAX_LEN), - "table_algorithm": TABLE_MASTER, + "table_algorithm": "TableMaster", "table_model_dir": table_model_dir, "table_char_dict_path": table_char_dict_path, "det_model_dir": det_model_dir, diff --git a/magic_pdf/para/para_split_v3.py b/magic_pdf/para/para_split_v3.py index 0ee2004a..237c4a5e 100644 --- a/magic_pdf/para/para_split_v3.py +++ b/magic_pdf/para/para_split_v3.py @@ -15,6 +15,9 @@ class ListLineTag: def __process_blocks(blocks): + # 对所有block预处理 + # 1.通过title和interline_equation将block分组 + # 2.bbox边界根据line信息重置 result = [] current_group = [] @@ -47,12 +50,16 @@ def __process_blocks(blocks): return result -def __is_list_block(block): +def __is_list_or_index_block(block): # 一个block如果是list block 应该同时满足以下特征 # 1.block内有多个line 2.block 内有多个line左侧顶格写 3.block内有多个line 右侧不顶格(狗牙状) # 1.block内有多个line 2.block 内有多个line左侧顶格写 3.多个line以endflag结尾 # 1.block内有多个line 2.block 内有多个line左侧顶格写 3.block内有多个line 左侧不顶格 - if len(block['lines']) >= 3: + + # index block 是一种特殊的list block + # 一个block如果是index block 应该同时满足以下特征 + # 1.block内有多个line 2.block 内有多个line两侧均顶格写 3.line的开头或者结尾均为数字 + if len(block['lines']) >= 2: first_line = block['lines'][0] line_height = first_line['bbox'][3] - first_line['bbox'][1] block_weight = block['bbox_fs'][2] - block['bbox_fs'][0] @@ -60,7 +67,19 @@ def __is_list_block(block): left_close_num = 0 left_not_close_num = 0 right_not_close_num = 0 + right_close_num = 0 lines_text_list = [] + + multiple_para_flag = False + last_line = block['lines'][-1] + # 如果首行左边不顶格而右边顶格,末行左边顶格而右边不顶格 (第一行可能可以右边不顶格) + if (first_line['bbox'][0] - block['bbox_fs'][0] > line_height / 2 and + # block['bbox_fs'][2] - first_line['bbox'][2] < line_height and + abs(last_line['bbox'][0] - block['bbox_fs'][0]) < line_height / 2 and + block['bbox_fs'][2] - last_line['bbox'][2] > line_height + ): + multiple_para_flag = True + for line in block['lines']: line_text = "" @@ -73,110 +92,118 @@ def __is_list_block(block): lines_text_list.append(line_text) # 计算line左侧顶格数量是否大于2,是否顶格用abs(block['bbox_fs'][0] - line['bbox'][0]) < line_height/2 来判断 - if abs(block['bbox_fs'][0] - line['bbox'][0]) < line_height/2: + if abs(block['bbox_fs'][0] - line['bbox'][0]) < line_height / 2: left_close_num += 1 elif line['bbox'][0] - block['bbox_fs'][0] > line_height: # logger.info(f"{line_text}, {block['bbox_fs']}, {line['bbox']}") left_not_close_num += 1 - # 计算右侧是否不顶格,拍脑袋用0.3block宽度做阈值 - closed_area = 0.3 * block_weight - # closed_area = 5 * line_height - if block['bbox_fs'][2] - line['bbox'][2] > closed_area: - right_not_close_num += 1 + # 计算右侧是否顶格 + if abs(block['bbox_fs'][2] - line['bbox'][2]) < line_height: + right_close_num += 1 + else: + # 右侧不顶格情况下是否有一段距离,拍脑袋用0.3block宽度做阈值 + closed_area = 0.3 * block_weight + # closed_area = 5 * line_height + if block['bbox_fs'][2] - line['bbox'][2] > closed_area: + right_not_close_num += 1 # 判断lines_text_list中的元素是否有超过80%都以LIST_END_FLAG结尾 line_end_flag = False + # 判断lines_text_list中的元素是否有超过80%都以数字开头或都以数字结尾 + line_num_flag = False + num_start_count = 0 + num_end_count = 0 + flag_end_count = 0 if len(lines_text_list) > 0: - num_end_count = 0 for line_text in lines_text_list: if len(line_text) > 0: if line_text[-1] in LIST_END_FLAG: - num_end_count += 1 - - if num_end_count / len(lines_text_list) >= 0.8: - line_end_flag = True - - if left_close_num >= 2 and (right_not_close_num >= 2 or line_end_flag or left_not_close_num >= 2): - for line in block['lines']: - if abs(block['bbox_fs'][0] - line['bbox'][0]) < line_height / 2: - line[ListLineTag.IS_LIST_START_LINE] = True - if abs(block['bbox_fs'][2] - line['bbox'][2]) > line_height: - line[ListLineTag.IS_LIST_END_LINE] = True - - return True - else: - return False - else: - return False - - -def __is_index_block(block): - # 一个block如果是index block 应该同时满足以下特征 - # 1.block内有多个line 2.block 内有多个line两侧均顶格写 3.line的开头或者结尾均为数字 - if len(block['lines']) >= 3: - first_line = block['lines'][0] - line_height = first_line['bbox'][3] - first_line['bbox'][1] - - left_close_num = 0 - right_close_num = 0 - - lines_text_list = [] - for line in block['lines']: - - # 计算line左侧顶格数量是否大于2,是否顶格用abs(block['bbox_fs'][0] - line['bbox'][0]) < line_height/2 来判断 - if abs(block['bbox_fs'][0] - line['bbox'][0]) < line_height / 2: - left_close_num += 1 - - # 计算右侧是否不顶格 - if abs(block['bbox_fs'][2] - line['bbox'][2]) < line_height / 2: - right_close_num += 1 - - line_text = "" - - for span in line['spans']: - span_type = span['type'] - if span_type == ContentType.Text: - line_text += span['content'].strip() - - lines_text_list.append(line_text) - - # 判断lines_text_list中的元素是否有超过80%都以数字开头或都以数字结尾 - line_num_flag = False - if len(lines_text_list) > 0: - num_start_count = 0 - num_end_count = 0 - for line_text in lines_text_list: - if len(line_text) > 0: + flag_end_count += 1 if line_text[0].isdigit(): num_start_count += 1 if line_text[-1].isdigit(): num_end_count += 1 + if flag_end_count / len(lines_text_list) >= 0.8: + line_end_flag = True + if num_start_count / len(lines_text_list) >= 0.8 or num_end_count / len(lines_text_list) >= 0.8: line_num_flag = True - if left_close_num >= 2 and right_close_num >= 2 and line_num_flag: + # 有的目录右侧不贴边, 目前认为左边或者右边有一边全贴边,且符合数字规则极为index + if ((left_close_num/len(block['lines']) >= 0.8 or right_close_num/len(block['lines']) >= 0.8) + and line_num_flag + ): for line in block['lines']: line[ListLineTag.IS_LIST_START_LINE] = True + return BlockType.Index - return True + elif left_close_num >= 2 and ( + right_not_close_num >= 2 or line_end_flag or left_not_close_num >= 2) and not multiple_para_flag: + # 处理一种特殊的没有缩进的list,所有行都贴左边,通过右边的空隙判断是否是item尾 + if left_close_num / len(block['lines']) > 0.9: + # 这种是每个item只有一行,且左边都贴边的短item list + if flag_end_count == 0 and right_close_num / len(block['lines']) < 0.5: + for line in block['lines']: + if abs(block['bbox_fs'][0] - line['bbox'][0]) < line_height / 2: + line[ListLineTag.IS_LIST_START_LINE] = True + # 这种是大部分line item 都有结束标识符的情况,按结束标识符区分不同item + elif line_end_flag: + for i, line in enumerate(block['lines']): + if lines_text_list[i][-1] in LIST_END_FLAG: + line[ListLineTag.IS_LIST_END_LINE] = True + if i + 1 < len(block['lines']): + block['lines'][i+1][ListLineTag.IS_LIST_START_LINE] = True + # line item基本没有结束标识符,而且也没有缩进,按右侧空隙判断哪些是item end + else: + line_start_flag = False + for i, line in enumerate(block['lines']): + if line_start_flag: + line[ListLineTag.IS_LIST_START_LINE] = True + line_start_flag = False + elif abs(block['bbox_fs'][2] - line['bbox'][2]) > line_height: + line[ListLineTag.IS_LIST_END_LINE] = True + line_start_flag = True + # 一种有缩进的特殊有序list,start line 左侧不贴边且以数字开头,end line 以 IS_LIST_END_LINE 结尾且数量和start line 一致 + elif num_start_count >= 2 and num_start_count == flag_end_count: # 简单一点先不考虑左侧不贴边的情况 + for i, line in enumerate(block['lines']): + if lines_text_list[i][0].isdigit(): + line[ListLineTag.IS_LIST_START_LINE] = True + if lines_text_list[i][-1] in LIST_END_FLAG: + line[ListLineTag.IS_LIST_END_LINE] = True + else: + # 正常有缩进的list处理 + for line in block['lines']: + if abs(block['bbox_fs'][0] - line['bbox'][0]) < line_height / 2: + line[ListLineTag.IS_LIST_START_LINE] = True + if abs(block['bbox_fs'][2] - line['bbox'][2]) > line_height: + line[ListLineTag.IS_LIST_END_LINE] = True + + return BlockType.List else: - return False + return BlockType.Text else: - return False + return BlockType.Text def __merge_2_text_blocks(block1, block2): if len(block1['lines']) > 0: first_line = block1['lines'][0] line_height = first_line['bbox'][3] - first_line['bbox'][1] - if abs(block1['bbox_fs'][0] - first_line['bbox'][0]) < line_height/2: + block1_weight = block1['bbox'][2] - block1['bbox'][0] + block2_weight = block2['bbox'][2] - block2['bbox'][0] + min_block_weight = min(block1_weight, block2_weight) + if abs(block1['bbox_fs'][0] - first_line['bbox'][0]) < line_height / 2: last_line = block2['lines'][-1] if len(last_line['spans']) > 0: last_span = last_line['spans'][-1] line_height = last_line['bbox'][3] - last_line['bbox'][1] - if abs(block2['bbox_fs'][2] - last_line['bbox'][2]) < line_height and not last_span['content'].endswith(LINE_STOP_FLAG): + if (abs(block2['bbox_fs'][2] - last_line['bbox'][2]) < line_height and + not last_span['content'].endswith(LINE_STOP_FLAG) and + # 两个block宽度差距超过2倍也不合并 + abs(block1_weight - block2_weight) < min_block_weight + ): if block1['page_num'] != block2['page_num']: for line in block1['lines']: for span in line['spans']: @@ -189,7 +216,6 @@ def __merge_2_text_blocks(block1, block2): def __merge_2_list_blocks(block1, block2): - if block1['page_num'] != block2['page_num']: for line in block1['lines']: for span in line['spans']: @@ -201,33 +227,47 @@ def __merge_2_list_blocks(block1, block2): return block1, block2 +def __is_list_group(text_blocks_group): + # list group的特征是一个group内的所有block都满足以下条件 + # 1.每个block都不超过3行 2. 每个block 的左边界都比较接近(逻辑简单点先不加这个规则) + for block in text_blocks_group: + if len(block['lines']) > 3: + return False + return True + + def __para_merge_page(blocks): page_text_blocks_groups = __process_blocks(blocks) for text_blocks_group in page_text_blocks_groups: if len(text_blocks_group) > 0: - # 需要先在合并前对所有block判断是否为list block + # 需要先在合并前对所有block判断是否为list or index block for block in text_blocks_group: - if __is_list_block(block): - block['type'] = BlockType.List - elif __is_index_block(block): - block['type'] = BlockType.Index + block_type = __is_list_or_index_block(block) + block['type'] = block_type + # logger.info(f"{block['type']}:{block}") if len(text_blocks_group) > 1: + + # 在合并前判断这个group 是否是一个 list group + is_list_group = __is_list_group(text_blocks_group) + # 倒序遍历 - for i in range(len(text_blocks_group)-1, -1, -1): + for i in range(len(text_blocks_group) - 1, -1, -1): current_block = text_blocks_group[i] # 检查是否有前一个块 if i - 1 >= 0: prev_block = text_blocks_group[i - 1] - if current_block['type'] == 'text' and prev_block['type'] == 'text': + if current_block['type'] == 'text' and prev_block['type'] == 'text' and not is_list_group: __merge_2_text_blocks(current_block, prev_block) - if current_block['type'] == BlockType.List and prev_block['type'] == BlockType.List: - __merge_2_list_blocks(current_block, prev_block) - if current_block['type'] == BlockType.Index and prev_block['type'] == BlockType.Index: + elif ( + (current_block['type'] == BlockType.List and prev_block['type'] == BlockType.List) or + (current_block['type'] == BlockType.Index and prev_block['type'] == BlockType.Index) + ): __merge_2_list_blocks(current_block, prev_block) + else: continue @@ -249,7 +289,7 @@ def para_split(pdf_info_dict, debug_mode=False): if __name__ == '__main__': - input_blocks = [{'type': 'text', 'bbox': [19, 79, 285, 95], 'lines': [{'bbox': [21.360000610351562, 81.50750732421875, 287.69000244140625, 93.62750244140625], 'spans': [{'bbox': [21.360000610351562, 81.62750244140625, 170.3000030517578, 93.62750244140625], 'content': '嘉和美康(688246)/计算机', 'type': 'text', 'score': 1.0}, {'bbox': [170.3000030517578, 81.62750244140625, 176.3000030517578, 93.62750244140625], 'content': ' ', 'type': 'text', 'score': 1.0}, {'bbox': [181.22000122070312, 81.50750732421875, 281.8052062988281, 93.50750732421875], 'content': '证券研究报告/公司点评', 'type': 'text', 'score': 1.0}, {'bbox': [281.69000244140625, 81.50750732421875, 287.69000244140625, 93.50750732421875], 'content': ' ', 'type': 'text', 'score': 1.0}], 'index': 0}], 'index': 0, 'page_num': 'page_0', 'bbox_fs': [21.360000610351562, 81.50750732421875, 287.69000244140625, 93.62750244140625]}, {'type': 'title', 'bbox': [18, 109, 124, 123], 'lines': [{'bbox': [21.360000610351562, 101.70799255371094, 98.47967529296875, 116.21743774414062], 'spans': [{'bbox': [21.360000610351562, 101.70799255371094, 98.47967529296875, 116.21743774414062], 'content': '[Table_Industry] ', 'type': 'text', 'score': 1.0}], 'index': 1}, {'bbox': [21.1200008392334, 110.3074951171875, 129.5640106201172, 122.3074951171875], 'spans': [{'bbox': [21.1200008392334, 110.3074951171875, 129.5640106201172, 122.3074951171875], 'content': '评级:买入(维持)', 'type': 'text', 'score': 1.0}], 'index': 2}], 'index': 1.5, 'page_num': 'page_0'}, {'type': 'text', 'bbox': [20, 126, 117, 137], 'lines': [{'bbox': [21.1200008392334, 127.40557861328125, 116.18000030517578, 136.40557861328125], 'spans': [{'bbox': [21.1200008392334, 127.40557861328125, 116.18000030517578, 136.40557861328125], 'content': '市场价格:16.62 元/股', 'type': 'text', 'score': 1.0}], 'index': 3}], 'index': 3, 'page_num': 'page_0', 'bbox_fs': [21.1200008392334, 127.40557861328125, 116.18000030517578, 136.40557861328125]}, {'type': 'text', 'bbox': [19, 144, 158, 172], 'lines': [{'bbox': [21.1200008392334, 144.1099853515625, 86.88600158691406, 156.50299072265625], 'spans': [{'bbox': [21.1200008392334, 146.005615234375, 84.33599853515625, 155.005615234375], 'content': '分析师:闻学臣', 'type': 'text', 'score': 1.0}, {'bbox': [84.38400268554688, 144.1099853515625, 86.88600158691406, 156.50299072265625], 'content': ' ', 'type': 'text', 'score': 1.0}], 'index': 4}, {'bbox': [21.1200008392334, 159.7099609375, 157.9219970703125, 172.10296630859375], 'spans': [{'bbox': [21.1200008392334, 161.6055908203125, 84.33599853515625, 170.6055908203125], 'content': '执业证书编号:', 'type': 'text', 'score': 1.0}, {'bbox': [84.50399780273438, 159.7099609375, 155.45095825195312, 172.10296630859375], 'content': 'S0740519090007', 'type': 'text', 'score': 1.0}, {'bbox': [155.4199981689453, 159.7099609375, 157.9219970703125, 172.10296630859375], 'content': ' ', 'type': 'text', 'score': 1.0}], 'index': 5}], 'index': 4.5, 'page_num': 'page_0', 'bbox_fs': [21.1200008392334, 144.1099853515625, 157.9219970703125, 172.10296630859375]}, {'type': 'text', 'bbox': [18, 194, 157, 241], 'lines': [{'bbox': [21.1200008392334, 193.86497497558594, 86.88600158691406, 206.23097229003906], 'spans': [{'bbox': [21.1200008392334, 195.80560302734375, 84.33599853515625, 204.80560302734375], 'content': '分析师:何柄谕', 'type': 'text', 'score': 1.0}, {'bbox': [84.38400268554688, 193.86497497558594, 86.88600158691406, 206.23097229003906], 'content': ' ', 'type': 'text', 'score': 1.0}], 'index': 6}, {'bbox': [21.1200008392334, 211.07000732421875, 157.9219970703125, 223.4630126953125], 'spans': [{'bbox': [21.1200008392334, 212.96563720703125, 84.33599853515625, 221.96563720703125], 'content': '执业证书编号:', 'type': 'text', 'score': 1.0}, {'bbox': [84.50399780273438, 211.07000732421875, 155.44796752929688, 223.4630126953125], 'content': 'S0740519090003', 'type': 'text', 'score': 1.0}, {'bbox': [155.4199981689453, 211.07000732421875, 157.9219970703125, 223.4630126953125], 'content': ' ', 'type': 'text', 'score': 1.0}], 'index': 7}, {'bbox': [21.1200008392334, 228.0649871826172, 126.84199523925781, 240.4309844970703], 'spans': [{'bbox': [21.1200008392334, 228.0649871826172, 43.73700714111328, 240.4309844970703], 'content': 'Email', 'type': 'text', 'score': 1.0}, {'bbox': [43.79999923706055, 230.005615234375, 52.79999923706055, 239.005615234375], 'content': ':', 'type': 'text', 'score': 1.0}, {'bbox': [52.68000030517578, 228.0649871826172, 124.41200256347656, 240.4309844970703], 'content': 'heby@zts.com.cn', 'type': 'text', 'score': 1.0}, {'bbox': [124.33999633789062, 228.0649871826172, 126.84199523925781, 240.4309844970703], 'content': ' ', 'type': 'text', 'score': 1.0}], 'index': 8}], 'index': 7, 'page_num': 'page_0', 'bbox_fs': [21.1200008392334, 193.86497497558594, 157.9219970703125, 240.4309844970703]}, {'type': 'table', 'bbox': [18, 338, 169, 418], 'blocks': [{'bbox': [18, 356, 169, 418], 'type': 'table_body', 'lines': [{'bbox': [18, 356, 169, 418], 'spans': [{'bbox': [18, 356, 169, 418], 'score': 0.8198961019515991, 'type': 'table', 'image_path': '4123619a2e8de87ebe695a4e7703d09d957670491c939b1050c96bbf4104210e.jpg'}]}]}, {'bbox': [19, 338, 70, 352], 'type': 'table_caption', 'lines': [{'bbox': [21.1200008392334, 335.9779968261719, 85.39967346191406, 350.4874267578125], 'spans': [{'bbox': [21.1200008392334, 335.9779968261719, 85.39967346191406, 350.4874267578125], 'content': '[Table_Profit] ', 'type': 'text', 'score': 1.0}]}]}], 'index': 9.5, 'page_num': 'page_0'}, {'type': 'image', 'bbox': [19, 426, 163, 558], 'blocks': [{'bbox': [21, 452, 163, 558], 'type': 'image_body', 'lines': [{'bbox': [21, 452, 163, 558], 'spans': [{'bbox': [21, 452, 163, 558], 'score': 0.9999651312828064, 'type': 'image', 'image_path': '0e63ab24cdc2ac4cb0c46bf1ff7b9f094c092b9c5707810cbc2b7e30964cf8a1.jpg'}]}]}, {'bbox': [19, 426, 160, 440], 'type': 'image_caption', 'lines': [{'bbox': [21.1200008392334, 427.8774719238281, 165.74000549316406, 439.8774719238281], 'spans': [{'bbox': [21.1200008392334, 427.8774719238281, 165.74000549316406, 439.8774719238281], 'content': '股价与行业-市场走势对比 ', 'type': 'text', 'score': 1.0}]}]}], 'index': 11.5, 'page_num': 'page_0'}, {'type': 'title', 'bbox': [20, 569, 70, 583], 'lines': [{'bbox': [21.1200008392334, 570.70751953125, 75.38400268554688, 582.70751953125], 'spans': [{'bbox': [21.1200008392334, 570.70751953125, 75.38400268554688, 582.70751953125], 'content': '相关报告 ', 'type': 'text', 'score': 1.0}], 'index': 13}], 'index': 13, 'page_num': 'page_0'}, {'type': 'text', 'bbox': [20, 586, 168, 629], 'lines': [{'bbox': [21.1200008392334, 585.9849853515625, 166.1840057373047, 598.3509521484375], 'spans': [{'bbox': [21.1200008392334, 585.9849853515625, 28.661998748779297, 598.3509521484375], 'content': '1 ', 'type': 'text', 'score': 1.0}, {'bbox': [30.239999771118164, 587.9255981445312, 83.76300048828125, 596.9255981445312], 'content': '《嘉和美康(', 'type': 'text', 'score': 1.0}, {'bbox': [83.78399658203125, 585.9849853515625, 113.72698211669922, 598.3509521484375], 'content': '688246', 'type': 'text', 'score': 1.0}, {'bbox': [113.77999877929688, 587.9255981445312, 131.3000030517578, 596.9255981445312], 'content': '):', 'type': 'text', 'score': 1.0}, {'bbox': [130.82000732421875, 585.9849853515625, 140.74400329589844, 598.3509521484375], 'content': '24', 'type': 'text', 'score': 1.0}, {'bbox': [140.74400329589844, 587.9255981445312, 151.94000244140625, 596.9255981445312], 'content': ' 年', 'type': 'text', 'score': 1.0}, {'bbox': [154.22000122070312, 585.9849853515625, 166.1840057373047, 598.3509521484375], 'content': 'Q1', 'type': 'text', 'score': 1.0}], 'index': 14}, {'bbox': [21.1200008392334, 603.525634765625, 165.1199951171875, 612.525634765625], 'spans': [{'bbox': [21.1200008392334, 603.525634765625, 165.1199951171875, 612.525634765625], 'content': '收入显著改善,医疗大模型产品落地', 'type': 'text', 'score': 1.0}], 'index': 15}, {'bbox': [21.1200008392334, 617.1849975585938, 50.62199783325195, 629.5509643554688], 'spans': [{'bbox': [21.1200008392334, 619.1256103515625, 48.119998931884766, 628.1256103515625], 'content': '良好》', 'type': 'text', 'score': 1.0}, {'bbox': [48.119998931884766, 617.1849975585938, 50.62199783325195, 629.5509643554688], 'content': ' ', 'type': 'text', 'score': 1.0}], 'index': 16}], 'index': 15, 'page_num': 'page_0', 'bbox_fs': [21.1200008392334, 585.9849853515625, 166.1840057373047, 629.5509643554688]}, {'type': 'text', 'bbox': [19, 648, 167, 677], 'lines': [{'bbox': [21.1200008392334, 648.385009765625, 166.21701049804688, 660.7509765625], 'spans': [{'bbox': [21.1200008392334, 648.385009765625, 28.662002563476562, 660.7509765625], 'content': '2 ', 'type': 'text', 'score': 1.0}, {'bbox': [30.1200008392334, 650.3256225585938, 83.51700592041016, 659.3256225585938], 'content': '《嘉和美康(', 'type': 'text', 'score': 1.0}, {'bbox': [83.54399871826172, 648.385009765625, 113.48698425292969, 660.7509765625], 'content': '688246', 'type': 'text', 'score': 1.0}, {'bbox': [113.54000091552734, 650.3256225585938, 166.21701049804688, 659.3256225585938], 'content': '):收入逐季', 'type': 'text', 'score': 1.0}], 'index': 17}, {'bbox': [21.1200008392334, 663.9849853515625, 153.6020050048828, 676.3509521484375], 'spans': [{'bbox': [21.1200008392334, 665.9255981445312, 111.12000274658203, 674.9255981445312], 'content': '度加速,继续加大医疗', 'type': 'text', 'score': 1.0}, {'bbox': [113.41999816894531, 663.9849853515625, 121.9219970703125, 676.3509521484375], 'content': 'AI', 'type': 'text', 'score': 1.0}, {'bbox': [121.9219970703125, 665.9255981445312, 151.10299682617188, 674.9255981445312], 'content': ' 投入》', 'type': 'text', 'score': 1.0}, {'bbox': [151.10000610351562, 663.9849853515625, 153.6020050048828, 676.3509521484375], 'content': ' ', 'type': 'text', 'score': 1.0}], 'index': 18}], 'index': 17.5, 'page_num': 'page_0', 'bbox_fs': [21.1200008392334, 648.385009765625, 166.21701049804688, 676.3509521484375]}, {'type': 'text', 'bbox': [19, 695, 167, 738], 'lines': [{'bbox': [21.1200008392334, 695.1849975585938, 166.21701049804688, 707.5509643554688], 'spans': [{'bbox': [21.1200008392334, 695.1849975585938, 28.661998748779297, 707.5509643554688], 'content': '3 ', 'type': 'text', 'score': 1.0}, {'bbox': [30.1200008392334, 697.1256103515625, 83.51700592041016, 706.1256103515625], 'content': '《嘉和美康(', 'type': 'text', 'score': 1.0}, {'bbox': [83.54399871826172, 695.1849975585938, 113.48698425292969, 707.5509643554688], 'content': '688246', 'type': 'text', 'score': 1.0}, {'bbox': [113.54000091552734, 697.1256103515625, 166.21701049804688, 706.1256103515625], 'content': '):回购彰显', 'type': 'text', 'score': 1.0}], 'index': 19}, {'bbox': [21.1200008392334, 710.7849731445312, 160.22000122070312, 723.1509399414062], 'spans': [{'bbox': [21.1200008392334, 712.7255859375, 138.1199951171875, 721.7255859375], 'content': '公司发展信心,公司加大医疗', 'type': 'text', 'score': 1.0}, {'bbox': [140.4199981689453, 710.7849731445312, 148.9219970703125, 723.1509399414062], 'content': 'AI', 'type': 'text', 'score': 1.0}, {'bbox': [148.9219970703125, 712.7255859375, 160.22000122070312, 721.7255859375], 'content': ' 投', 'type': 'text', 'score': 1.0}], 'index': 20}, {'bbox': [21.1200008392334, 726.4049682617188, 41.62199783325195, 738.7709350585938], 'spans': [{'bbox': [21.1200008392334, 728.3455810546875, 39.12000274658203, 737.3455810546875], 'content': '入》', 'type': 'text', 'score': 1.0}, {'bbox': [39.119998931884766, 726.4049682617188, 41.62199783325195, 738.7709350585938], 'content': ' ', 'type': 'text', 'score': 1.0}], 'index': 21}], 'index': 20, 'page_num': 'page_0', 'bbox_fs': [21.1200008392334, 695.1849975585938, 166.21701049804688, 738.7709350585938]}, {'type': 'text', 'bbox': [427, 80, 506, 94], 'lines': [{'bbox': [429.54998779296875, 81.50750732421875, 509.739990234375, 93.50750732421875], 'spans': [{'bbox': [429.54998779296875, 81.50750732421875, 503.8600158691406, 93.50750732421875], 'content': '2024 年8 月28 日', 'type': 'text', 'score': 1.0}, {'bbox': [503.739990234375, 81.50750732421875, 509.739990234375, 93.50750732421875], 'content': ' ', 'type': 'text', 'score': 1.0}], 'index': 22}], 'index': 22, 'page_num': 'page_0', 'bbox_fs': [429.54998779296875, 81.50750732421875, 509.739990234375, 93.50750732421875]}, {'type': 'table', 'bbox': [184, 108, 568, 273], 'blocks': [{'bbox': [184, 124, 568, 249], 'type': 'table_body', 'lines': [{'bbox': [184, 124, 568, 249], 'spans': [{'bbox': [184, 124, 568, 249], 'score': 0.9999539852142334, 'type': 'table', 'image_path': 'feabef6394c4fd70ba64aece3701cd1fc49a0b7deb4ea0693dd63131f182fb9c.jpg'}]}]}, {'bbox': [184, 108, 295, 122], 'type': 'table_caption', 'lines': [{'bbox': [186.5, 110.3074951171875, 294.9320068359375, 122.3074951171875], 'spans': [{'bbox': [186.5, 110.3074951171875, 294.9320068359375, 122.3074951171875], 'content': '公司盈利预测及估值', 'type': 'text', 'score': 1.0}]}]}, {'bbox': [184, 262, 344, 273], 'type': 'table_footnote', 'lines': [{'bbox': [186.5, 262.17498779296875, 343.1300048828125, 274.5409851074219], 'spans': [{'bbox': [186.5, 264.1156005859375, 213.5, 273.1156005859375], 'content': '备注:', 'type': 'text', 'score': 1.0}, {'bbox': [213.52999877929688, 264.1156005859375, 240.52999877929688, 273.1156005859375], 'content': '股价为', 'type': 'text', 'score': 1.0}, {'bbox': [242.80999755859375, 262.17498779296875, 262.8139953613281, 274.5409851074219], 'content': '2024', 'type': 'text', 'score': 1.0}, {'bbox': [262.8139953613281, 264.1156005859375, 274.1300048828125, 273.1156005859375], 'content': ' 年', 'type': 'text', 'score': 1.0}, {'bbox': [276.4100036621094, 262.17498779296875, 281.41400146484375, 274.5409851074219], 'content': '8', 'type': 'text', 'score': 1.0}, {'bbox': [281.41400146484375, 264.1156005859375, 292.6099853515625, 273.1156005859375], 'content': ' 月', 'type': 'text', 'score': 1.0}, {'bbox': [294.8900146484375, 262.17498779296875, 304.93402099609375, 274.5409851074219], 'content': '27', 'type': 'text', 'score': 1.0}, {'bbox': [304.93402099609375, 264.1156005859375, 343.1300048828125, 273.1156005859375], 'content': ' 日收盘价', 'type': 'text', 'score': 1.0}]}]}], 'index': 24, 'page_num': 'page_0'}, {'type': 'title', 'bbox': [180, 285, 230, 300], 'lines': [{'bbox': [186.5, 277.7750244140625, 189.0019989013672, 290.1410217285156], 'spans': [{'bbox': [186.5, 277.7750244140625, 189.0019989013672, 290.1410217285156], 'content': ' ', 'type': 'text', 'score': 1.0}], 'index': 26}, {'bbox': [180.86000061035156, 280.41796875, 183.79568481445312, 294.9273986816406], 'spans': [{'bbox': [180.86000061035156, 280.41796875, 183.79568481445312, 294.9273986816406], 'content': ' ', 'type': 'text', 'score': 1.0}], 'index': 27}, {'bbox': [180.86000061035156, 287.09747314453125, 235.1300048828125, 299.09747314453125], 'spans': [{'bbox': [180.86000061035156, 287.09747314453125, 235.1300048828125, 299.09747314453125], 'content': '投资要点 ', 'type': 'text', 'score': 1.0}], 'index': 28}], 'index': 27, 'page_num': 'page_0'}, {'type': 'text', 'bbox': [198, 302, 578, 331], 'lines': [{'bbox': [201.88999938964844, 302.3030090332031, 575.02001953125, 315.988037109375], 'spans': [{'bbox': [201.88999938964844, 304.45062255859375, 292.0099792480469, 314.41064453125], 'content': '投资事件:公司发布', 'type': 'text', 'score': 1.0}, {'bbox': [294.6499938964844, 302.3030090332031, 316.8507995605469, 315.988037109375], 'content': '2024', 'type': 'text', 'score': 1.0}, {'bbox': [316.8507995605469, 304.45062255859375, 429.3785705566406, 314.41064453125], 'content': ' 年中报:营业收入规模达', 'type': 'text', 'score': 1.0}, {'bbox': [432.07000732421875, 302.3030090332031, 451.5318298339844, 315.988037109375], 'content': '3.00', 'type': 'text', 'score': 1.0}, {'bbox': [451.5318298339844, 304.45062255859375, 524.1190795898438, 314.41064453125], 'content': ' 亿元,同比增长', 'type': 'text', 'score': 1.0}, {'bbox': [525, 303, 556, 314], 'score': 0.82, 'content': '2.92\\%', 'type': 'inline_equation'}, {'bbox': [555.0999755859375, 304.45062255859375, 575.02001953125, 314.41064453125], 'content': ',归', 'type': 'text', 'score': 1.0}], 'index': 29}, {'bbox': [201.88999938964844, 317.9029846191406, 329.118896484375, 331.6676940917969], 'spans': [{'bbox': [201.88999938964844, 320.05059814453125, 271.7195739746094, 330.0106201171875], 'content': '母净利润为亏损', 'type': 'text', 'score': 1.0}, {'bbox': [274.3699951171875, 317.9029846191406, 293.69873046875, 331.5880126953125], 'content': '0.27', 'type': 'text', 'score': 1.0}, {'bbox': [293.69873046875, 320.05059814453125, 326.31951904296875, 330.0106201171875], 'content': ' 亿元。', 'type': 'text', 'score': 1.0}, {'bbox': [326.3500061035156, 317.9527893066406, 329.118896484375, 331.6676940917969], 'content': ' ', 'type': 'text', 'score': 1.0}], 'index': 30}], 'index': 29.5, 'page_num': 'page_0', 'bbox_fs': [201.88999938964844, 302.3030090332031, 575.02001953125, 331.6676940917969]}, {'type': 'text', 'bbox': [199, 349, 576, 425], 'lines': [{'bbox': [201.88999938964844, 351.2506103515625, 574.9908447265625, 361.21063232421875], 'spans': [{'bbox': [201.88999938964844, 351.2506103515625, 574.9908447265625, 361.21063232421875], 'content': '收入小幅增长,毛利率改善。报告期内,公司医疗临床业务、医疗数据业务等业务板', 'type': 'text', 'score': 1.0}], 'index': 31}, {'bbox': [201.88999938964844, 364.7029724121094, 577.1592407226562, 378.38800048828125], 'spans': [{'bbox': [201.88999938964844, 366.8505859375, 331.8081970214844, 376.81060791015625], 'content': '块平稳发展,整体收入规模达', 'type': 'text', 'score': 1.0}, {'bbox': [334.3900146484375, 364.7029724121094, 353.71875, 378.38800048828125], 'content': '3.00', 'type': 'text', 'score': 1.0}, {'bbox': [353.71875, 366.8505859375, 426.17950439453125, 376.81060791015625], 'content': ' 亿元,同比增长', 'type': 'text', 'score': 1.0}, {'bbox': [427, 365, 457, 377], 'score': 0.92, 'content': '2.92\\%', 'type': 'inline_equation'}, {'bbox': [457.17999267578125, 366.8505859375, 577.1592407226562, 376.81060791015625], 'content': ',整体收入实现平稳增长。', 'type': 'text', 'score': 1.0}], 'index': 32}, {'bbox': [201.88999938964844, 382.4505920410156, 580.0416259765625, 392.4106140136719], 'spans': [{'bbox': [201.88999938964844, 382.4505920410156, 580.0416259765625, 392.4106140136719], 'content': '由于公司优化产品结构,改进实施交付管理,公司业务毛利空间有所提升。报告期内,', 'type': 'text', 'score': 1.0}], 'index': 33}, {'bbox': [201.88999938964844, 395.9229736328125, 574.8645629882812, 409.6080017089844], 'spans': [{'bbox': [201.88999938964844, 398.0705871582031, 291.7491149902344, 408.0306091308594], 'content': '公司综合毛利率达到', 'type': 'text', 'score': 1.0}, {'bbox': [293, 397, 328, 409], 'score': 0.89, 'content': '48.03\\%', 'type': 'inline_equation'}, {'bbox': [328.2699890136719, 398.0705871582031, 386.6952819824219, 408.0306091308594], 'content': ',去年同期为', 'type': 'text', 'score': 1.0}, {'bbox': [388, 397, 423, 409], 'score': 0.89, 'content': '45.52\\%', 'type': 'inline_equation'}, {'bbox': [423.30999755859375, 398.0705871582031, 471.7752990722656, 408.0306091308594], 'content': ',同比提升', 'type': 'text', 'score': 1.0}, {'bbox': [474.3399963378906, 395.9229736328125, 493.80181884765625, 409.6080017089844], 'content': '2.51', 'type': 'text', 'score': 1.0}, {'bbox': [493.80181884765625, 398.0705871582031, 574.8645629882812, 408.0306091308594], 'content': ' 个百分点,公司毛', 'type': 'text', 'score': 1.0}], 'index': 34}, {'bbox': [201.88999938964844, 411.5229797363281, 279.6589050292969, 425.2080078125], 'spans': [{'bbox': [201.88999938964844, 413.67059326171875, 271.7195739746094, 423.630615234375], 'content': '利率明显改善。', 'type': 'text', 'score': 1.0}, {'bbox': [271.7300109863281, 411.5229797363281, 279.6589050292969, 425.2080078125], 'content': ' ', 'type': 'text', 'score': 1.0}], 'index': 35}], 'index': 33, 'page_num': 'page_0'}, {'type': 'text', 'bbox': [199, 427, 577, 503], 'lines': [{'bbox': [201.88999938964844, 429.2705993652344, 574.9743041992188, 439.2306213378906], 'spans': [{'bbox': [201.88999938964844, 429.2705993652344, 574.9743041992188, 439.2306213378906], 'content': '降本增效成效显著,管理、销售费用率下降。报告期内,公司注重内控管理、人员能', 'type': 'text', 'score': 1.0}], 'index': 36}, {'bbox': [201.88999938964844, 442.7229919433594, 575.1400146484375, 456.40802001953125], 'spans': [{'bbox': [201.88999938964844, 444.87060546875, 530.7092895507812, 454.83062744140625], 'content': '效提升,加强管理方式优化及费用控制,公司运营管理方面降本增效明显。', 'type': 'text', 'score': 1.0}, {'bbox': [530.3800048828125, 442.7229919433594, 552.600830078125, 456.40802001953125], 'content': '2024', 'type': 'text', 'score': 1.0}, {'bbox': [552.600830078125, 444.87060546875, 575.1400146484375, 454.83062744140625], 'content': ' 年上', 'type': 'text', 'score': 1.0}], 'index': 37}, {'bbox': [201.88999938964844, 458.3229675292969, 575.1334838867188, 472.00799560546875], 'spans': [{'bbox': [201.88999938964844, 460.4705810546875, 310.71295166015625, 470.43060302734375], 'content': '半年,公司销售费用率为', 'type': 'text', 'score': 1.0}, {'bbox': [312, 459, 348, 471], 'score': 0.91, 'content': '16.39\\%', 'type': 'inline_equation'}, {'bbox': [347.3500061035156, 460.4705810546875, 406.1438293457031, 470.43060302734375], 'content': ',去年同期为', 'type': 'text', 'score': 1.0}, {'bbox': [407, 459, 443, 471], 'score': 0.9, 'content': '17.57\\%', 'type': 'inline_equation'}, {'bbox': [442.75, 460.4705810546875, 501.5438232421875, 470.43060302734375], 'content': ',同比下降个', 'type': 'text', 'score': 1.0}, {'bbox': [504.2200012207031, 458.3229675292969, 523.5487670898438, 472.00799560546875], 'content': '1.18', 'type': 'text', 'score': 1.0}, {'bbox': [523.5487670898438, 460.4705810546875, 575.1334838867188, 470.43060302734375], 'content': ' 百分点;管', 'type': 'text', 'score': 1.0}], 'index': 38}, {'bbox': [201.88999938964844, 473.9229736328125, 575.0936279296875, 487.6080017089844], 'spans': [{'bbox': [201.88999938964844, 476.0705871582031, 251.79959106445312, 486.0306091308594], 'content': '理费用率为', 'type': 'text', 'score': 1.0}, {'bbox': [253, 474, 288, 487], 'score': 0.89, 'content': '16.21\\%', 'type': 'inline_equation'}, {'bbox': [288.2900085449219, 476.0705871582031, 346.8248596191406, 486.0306091308594], 'content': ',去年同期为', 'type': 'text', 'score': 1.0}, {'bbox': [348, 474, 384, 487], 'score': 0.89, 'content': '17.79\\%', 'type': 'inline_equation'}, {'bbox': [383.3500061035156, 476.0705871582031, 431.9348449707031, 486.0306091308594], 'content': ',同比下降', 'type': 'text', 'score': 1.0}, {'bbox': [434.5899963378906, 473.9229736328125, 453.9187316894531, 487.6080017089844], 'content': '1.58', 'type': 'text', 'score': 1.0}, {'bbox': [453.9187316894531, 476.0705871582031, 575.0936279296875, 486.0306091308594], 'content': ' 个百分点。公司管理费用率', 'type': 'text', 'score': 1.0}], 'index': 39}, {'bbox': [201.88999938964844, 489.5727844238281, 434.7189025878906, 503.2876892089844], 'spans': [{'bbox': [201.88999938964844, 491.67059326171875, 431.7367858886719, 501.630615234375], 'content': '及销售费用率均实现下降,公司运营效率明显提升。', 'type': 'text', 'score': 1.0}, {'bbox': [431.95001220703125, 489.5727844238281, 434.7189025878906, 503.2876892089844], 'content': ' ', 'type': 'text', 'score': 1.0}], 'index': 40}], 'index': 38, 'page_num': 'page_0'}, {'type': 'text', 'bbox': [199, 505, 577, 628], 'lines': [{'bbox': [201.88999938964844, 505.1727600097656, 575.0682983398438, 518.8876953125], 'spans': [{'bbox': [201.88999938964844, 507.27056884765625, 241.9491424560547, 517.2305908203125], 'content': '公司加大', 'type': 'text', 'score': 1.0}, {'bbox': [245.2100067138672, 505.1727600097656, 255.1788787841797, 518.8876953125], 'content': 'AI', 'type': 'text', 'score': 1.0}, {'bbox': [255.1788787841797, 507.27056884765625, 328.44818115234375, 517.2305908203125], 'content': ' 投入力度,医疗', 'type': 'text', 'score': 1.0}, {'bbox': [331.75, 505.1727600097656, 341.7189025878906, 518.8876953125], 'content': 'AI', 'type': 'text', 'score': 1.0}, {'bbox': [341.7189025878906, 507.27056884765625, 575.0682983398438, 517.2305908203125], 'content': ' 产品落地情况良好。公司继续加大研发投入力度,尤', 'type': 'text', 'score': 1.0}], 'index': 41}, {'bbox': [201.88999938964844, 520.7230224609375, 575.057861328125, 534.4080200195312], 'spans': [{'bbox': [201.88999938964844, 522.87060546875, 241.83958435058594, 532.8306274414062], 'content': '其是医疗', 'type': 'text', 'score': 1.0}, {'bbox': [244.97000122070312, 520.7230224609375, 254.45889282226562, 534.4080200195312], 'content': 'AI', 'type': 'text', 'score': 1.0}, {'bbox': [254.45889282226562, 522.87060546875, 407.5176696777344, 532.8306274414062], 'content': ' 投入力度。报告期内,公司新申请', 'type': 'text', 'score': 1.0}, {'bbox': [410.8299865722656, 520.7230224609375, 421.8876953125, 534.4080200195312], 'content': '26', 'type': 'text', 'score': 1.0}, {'bbox': [421.8876953125, 522.87060546875, 575.057861328125, 532.8306274414062], 'content': ' 项发明专利,主要集中在医疗数据', 'type': 'text', 'score': 1.0}], 'index': 42}, {'bbox': [201.88999938964844, 536.322998046875, 574.8896484375, 550.0079956054688], 'spans': [{'bbox': [201.88999938964844, 538.4705810546875, 231.77001953125, 548.4306030273438], 'content': '利用和', 'type': 'text', 'score': 1.0}, {'bbox': [234.77000427246094, 536.322998046875, 244.13890075683594, 550.0079956054688], 'content': 'AI', 'type': 'text', 'score': 1.0}, {'bbox': [244.13890075683594, 538.4705810546875, 306.98907470703125, 548.4306030273438], 'content': ' 领域,并获得', 'type': 'text', 'score': 1.0}, {'bbox': [309.8900146484375, 536.322998046875, 315.4277648925781, 550.0079956054688], 'content': '1', 'type': 'text', 'score': 1.0}, {'bbox': [315.4277648925781, 538.4705810546875, 368.31951904296875, 548.4306030273438], 'content': ' 项核心技术', 'type': 'text', 'score': 1.0}, {'bbox': [368.3500061035156, 538.013427734375, 371.6667785644531, 549.140625], 'content': '“', 'type': 'text', 'score': 1.0}, {'bbox': [371.7099914550781, 538.4705810546875, 521.548095703125, 548.4306030273438], 'content': '大模型辅助电子病历自动生成技术', 'type': 'text', 'score': 1.0}, {'bbox': [521.6199951171875, 538.013427734375, 524.936767578125, 549.140625], 'content': '”', 'type': 'text', 'score': 1.0}, {'bbox': [524.97998046875, 538.4705810546875, 574.8896484375, 548.4306030273438], 'content': '。依托公司', 'type': 'text', 'score': 1.0}], 'index': 43}, {'bbox': [201.88999938964844, 551.9229736328125, 574.925048828125, 565.6080322265625], 'spans': [{'bbox': [201.88999938964844, 551.9229736328125, 211.25889587402344, 565.6080322265625], 'content': 'AI', 'type': 'text', 'score': 1.0}, {'bbox': [211.25889587402344, 554.0706176757812, 332.65252685546875, 564.0306396484375], 'content': ' 技术的积累,公司推出医疗', 'type': 'text', 'score': 1.0}, {'bbox': [335.2300109863281, 551.9229736328125, 344.5989074707031, 565.6080322265625], 'content': 'AI', 'type': 'text', 'score': 1.0}, {'bbox': [344.5989074707031, 554.0706176757812, 574.925048828125, 564.0306396484375], 'content': ' 应用开发平台,打造全院智慧化服务接入底座,实现', 'type': 'text', 'score': 1.0}], 'index': 44}, {'bbox': [201.88999938964844, 567.552978515625, 574.8896484375, 581.238037109375], 'spans': [{'bbox': [201.88999938964844, 569.7006225585938, 291.7491149902344, 579.66064453125], 'content': '多技术框架、多业务', 'type': 'text', 'score': 1.0}, {'bbox': [294.4100036621094, 567.552978515625, 303.7789001464844, 581.238037109375], 'content': 'AI', 'type': 'text', 'score': 1.0}, {'bbox': [303.7789001464844, 569.7006225585938, 356.19952392578125, 579.66064453125], 'content': ' 应用接入。', 'type': 'text', 'score': 1.0}, {'bbox': [356.3500061035156, 567.552978515625, 378.5508117675781, 581.238037109375], 'content': '2024', 'type': 'text', 'score': 1.0}, {'bbox': [378.5508117675781, 569.7006225585938, 391.0299987792969, 579.66064453125], 'content': ' 年', 'type': 'text', 'score': 1.0}, {'bbox': [393.54998779296875, 567.552978515625, 399.0877380371094, 581.238037109375], 'content': '7', 'type': 'text', 'score': 1.0}, {'bbox': [399.0877380371094, 569.7006225585938, 531.5186157226562, 579.66064453125], 'content': ' 月,公司与北医三院联合发布', 'type': 'text', 'score': 1.0}, {'bbox': [531.5800170898438, 569.2434692382812, 534.8967895507812, 580.3706665039062], 'content': '“', 'type': 'text', 'score': 1.0}, {'bbox': [534.9400024414062, 569.7006225585938, 574.8896484375, 579.66064453125], 'content': '三生大模', 'type': 'text', 'score': 1.0}], 'index': 45}, {'bbox': [201.88999938964844, 583.1529541015625, 575.1400146484375, 596.8380126953125], 'spans': [{'bbox': [201.88999938964844, 585.3005981445312, 211.85000610351562, 595.2606201171875], 'content': '型', 'type': 'text', 'score': 1.0}, {'bbox': [211.85000610351562, 584.8434448242188, 215.16676330566406, 595.9706420898438], 'content': '”', 'type': 'text', 'score': 1.0}, {'bbox': [215.2100067138672, 585.3005981445312, 540.5800170898438, 595.2606201171875], 'content': ',以大模型为底座的多业务场景得到落地验证并且应用效果良好,比如新型', 'type': 'text', 'score': 1.0}, {'bbox': [543.219970703125, 583.1529541015625, 552.5888671875, 596.8380126953125], 'content': 'AI', 'type': 'text', 'score': 1.0}, {'bbox': [552.5888671875, 585.3005981445312, 575.1400146484375, 595.2606201171875], 'content': ' 产品', 'type': 'text', 'score': 1.0}], 'index': 46}, {'bbox': [201.88999938964844, 600.9005737304688, 575.1082763671875, 610.860595703125], 'spans': [{'bbox': [201.88999938964844, 600.9005737304688, 575.1082763671875, 610.860595703125], 'content': '可以将医务人员曾经数小时的病历书写工作缩减至半小时内完成,大幅提升书写内容', 'type': 'text', 'score': 1.0}], 'index': 47}, {'bbox': [201.88999938964844, 614.4027709960938, 304.618896484375, 628.1177368164062], 'spans': [{'bbox': [201.88999938964844, 616.5006103515625, 301.7195129394531, 626.4606323242188], 'content': '的准确率及工作效率。', 'type': 'text', 'score': 1.0}, {'bbox': [301.8500061035156, 614.4027709960938, 304.618896484375, 628.1177368164062], 'content': ' ', 'type': 'text', 'score': 1.0}], 'index': 48}], 'index': 44.5, 'page_num': 'page_0'}, {'type': 'text', 'bbox': [200, 646, 577, 690], 'lines': [{'bbox': [201.88999938964844, 645.552978515625, 574.8973999023438, 659.238037109375], 'spans': [{'bbox': [201.88999938964844, 647.7006225585938, 310.00994873046875, 657.66064453125], 'content': '投资建议:我们预计公司', 'type': 'text', 'score': 1.0}, {'bbox': [312.5299987792969, 645.552978515625, 384.72003173828125, 659.238037109375], 'content': '2024/2025/2026', 'type': 'text', 'score': 1.0}, {'bbox': [384.72003173828125, 647.7006225585938, 447.2890625, 657.66064453125], 'content': ' 年收入分别为', 'type': 'text', 'score': 1.0}, {'bbox': [449.8299865722656, 645.552978515625, 526.9088745117188, 659.238037109375], 'content': '9.03/11.48/14.47 ', 'type': 'text', 'score': 1.0}, {'bbox': [526.9000244140625, 647.7006225585938, 574.8973999023438, 657.66064453125], 'content': '亿元,净利', 'type': 'text', 'score': 1.0}], 'index': 49}, {'bbox': [201.88999938964844, 661.1529541015625, 574.98876953125, 674.8380126953125], 'spans': [{'bbox': [201.88999938964844, 663.3005981445312, 241.7300262451172, 673.2606201171875], 'content': '润分别为', 'type': 'text', 'score': 1.0}, {'bbox': [241.85000610351562, 661.1529541015625, 311.3388977050781, 674.8380126953125], 'content': ' 0.95/1.20/1.60 ', 'type': 'text', 'score': 1.0}, {'bbox': [311.3299865722656, 663.3005981445312, 361.239501953125, 673.2606201171875], 'content': '亿元,对应', 'type': 'text', 'score': 1.0}, {'bbox': [364.989990234375, 661.1529541015625, 378.35333251953125, 674.8380126953125], 'content': 'PE', 'type': 'text', 'score': 1.0}, {'bbox': [378.35333251953125, 663.3005981445312, 411.8995361328125, 673.2606201171875], 'content': ' 分别为', 'type': 'text', 'score': 1.0}, {'bbox': [411.9100036621094, 661.1529541015625, 481.42193603515625, 674.8380126953125], 'content': ' 24.1/19.0/14.3', 'type': 'text', 'score': 1.0}, {'bbox': [481.42193603515625, 663.3005981445312, 574.98876953125, 673.2606201171875], 'content': ' 倍。考虑公司业绩高', 'type': 'text', 'score': 1.0}], 'index': 50}, {'bbox': [201.88999938964844, 676.802734375, 431.35888671875, 690.5177001953125], 'spans': [{'bbox': [201.88999938964844, 678.9005737304688, 371.7577209472656, 688.860595703125], 'content': '增长以及估值处于较低水平,给予公司', 'type': 'text', 'score': 1.0}, {'bbox': [371.8299865722656, 678.4434204101562, 375.1467590332031, 689.5706176757812], 'content': '“', 'type': 'text', 'score': 1.0}, {'bbox': [375.19000244140625, 678.9005737304688, 395.1099853515625, 688.860595703125], 'content': '买入', 'type': 'text', 'score': 1.0}, {'bbox': [395.1099853515625, 678.4434204101562, 398.4267578125, 689.5706176757812], 'content': '”', 'type': 'text', 'score': 1.0}, {'bbox': [398.4700012207031, 678.9005737304688, 428.45953369140625, 688.860595703125], 'content': '评级。', 'type': 'text', 'score': 1.0}, {'bbox': [428.5899963378906, 676.802734375, 431.35888671875, 690.5177001953125], 'content': ' ', 'type': 'text', 'score': 1.0}], 'index': 51}], 'index': 50, 'page_num': 'page_0'}, {'type': 'text', 'bbox': [200, 708, 404, 721], 'lines': [{'bbox': [201.88999938964844, 708.0027465820312, 404.9588928222656, 721.7177124023438], 'spans': [{'bbox': [201.88999938964844, 710.1005859375, 402.00811767578125, 720.0606079101562], 'content': '风险提示:业务发展不及预期,政策推进缓慢', 'type': 'text', 'score': 1.0}, {'bbox': [402.19000244140625, 708.0027465820312, 404.9588928222656, 721.7177124023438], 'content': ' ', 'type': 'text', 'score': 1.0}], 'index': 52}], 'index': 52, 'page_num': 'page_0'}] + input_blocks = [] # 调用函数 groups = __process_blocks(input_blocks) for group_index, group in enumerate(groups): diff --git a/magic_pdf/pdf_parse_by_ocr.py b/magic_pdf/pdf_parse_by_ocr.py index 0686d59e..ca2f394b 100644 --- a/magic_pdf/pdf_parse_by_ocr.py +++ b/magic_pdf/pdf_parse_by_ocr.py @@ -1,3 +1,5 @@ +from magic_pdf.config.enums import SupportedPdfParseMethod +from magic_pdf.data.dataset import PymuDocDataset from magic_pdf.pdf_parse_union_core_v2 import pdf_parse_union @@ -8,10 +10,11 @@ def parse_pdf_by_ocr(pdf_bytes, end_page_id=None, debug_mode=False, ): - return pdf_parse_union(pdf_bytes, + dataset = PymuDocDataset(pdf_bytes) + return pdf_parse_union(dataset, model_list, imageWriter, - "ocr", + SupportedPdfParseMethod.OCR, start_page_id=start_page_id, end_page_id=end_page_id, debug_mode=debug_mode, diff --git a/magic_pdf/pdf_parse_by_txt.py b/magic_pdf/pdf_parse_by_txt.py index bd8e202d..bae800f4 100644 --- a/magic_pdf/pdf_parse_by_txt.py +++ b/magic_pdf/pdf_parse_by_txt.py @@ -1,3 +1,5 @@ +from magic_pdf.config.enums import SupportedPdfParseMethod +from magic_pdf.data.dataset import PymuDocDataset from magic_pdf.pdf_parse_union_core_v2 import pdf_parse_union @@ -9,10 +11,11 @@ def parse_pdf_by_txt( end_page_id=None, debug_mode=False, ): - return pdf_parse_union(pdf_bytes, + dataset = PymuDocDataset(pdf_bytes) + return pdf_parse_union(dataset, model_list, imageWriter, - "txt", + SupportedPdfParseMethod.TXT, start_page_id=start_page_id, end_page_id=end_page_id, debug_mode=debug_mode, diff --git a/magic_pdf/pdf_parse_union_core_v2.py b/magic_pdf/pdf_parse_union_core_v2.py index f28e0d3c..4c3a6d00 100644 --- a/magic_pdf/pdf_parse_union_core_v2.py +++ b/magic_pdf/pdf_parse_union_core_v2.py @@ -1,13 +1,14 @@ +import copy import os import statistics import time - -from loguru import logger - from typing import List import torch +from loguru import logger +from magic_pdf.config.enums import SupportedPdfParseMethod +from magic_pdf.data.dataset import Dataset, PageableData from magic_pdf.libs.clean_memory import clean_memory from magic_pdf.libs.commons import fitz, get_delta_time from magic_pdf.libs.config_reader import get_local_layoutreader_model_dir @@ -15,31 +16,39 @@ from magic_pdf.libs.convert_utils import dict_to_list from magic_pdf.libs.drop_reason import DropReason from magic_pdf.libs.hash_utils import compute_md5 from magic_pdf.libs.local_math import float_equal -from magic_pdf.libs.ocr_content_type import ContentType +from magic_pdf.libs.ocr_content_type import ContentType, BlockType from magic_pdf.model.magic_model import MagicModel from magic_pdf.para.para_split_v3 import para_split from magic_pdf.pre_proc.citationmarker_remove import remove_citation_marker -from magic_pdf.pre_proc.construct_page_dict import ocr_construct_page_component_v2 +from magic_pdf.pre_proc.construct_page_dict import \ + ocr_construct_page_component_v2 from magic_pdf.pre_proc.cut_image import ocr_cut_image_and_table -from magic_pdf.pre_proc.equations_replace import remove_chars_in_text_blocks, replace_equations_in_textblock, \ - combine_chars_to_pymudict -from magic_pdf.pre_proc.ocr_detect_all_bboxes import ocr_prepare_bboxes_for_layout_split_v2 -from magic_pdf.pre_proc.ocr_dict_merge import fill_spans_in_blocks, fix_block_spans, fix_discarded_block -from magic_pdf.pre_proc.ocr_span_list_modify import remove_overlaps_min_spans, get_qa_need_list_v2, \ - remove_overlaps_low_confidence_spans -from magic_pdf.pre_proc.resolve_bbox_conflict import check_useful_block_horizontal_overlap +from magic_pdf.pre_proc.equations_replace import ( + combine_chars_to_pymudict, remove_chars_in_text_blocks, + replace_equations_in_textblock) +from magic_pdf.pre_proc.ocr_detect_all_bboxes import \ + ocr_prepare_bboxes_for_layout_split_v2 +from magic_pdf.pre_proc.ocr_dict_merge import (fill_spans_in_blocks, + fix_block_spans, + fix_discarded_block, fix_block_spans_v2) +from magic_pdf.pre_proc.ocr_span_list_modify import ( + get_qa_need_list_v2, remove_overlaps_low_confidence_spans, + remove_overlaps_min_spans) +from magic_pdf.pre_proc.resolve_bbox_conflict import \ + check_useful_block_horizontal_overlap def remove_horizontal_overlap_block_which_smaller(all_bboxes): useful_blocks = [] for bbox in all_bboxes: - useful_blocks.append({ - "bbox": bbox[:4] - }) - is_useful_block_horz_overlap, smaller_bbox, bigger_bbox = check_useful_block_horizontal_overlap(useful_blocks) + useful_blocks.append({'bbox': bbox[:4]}) + is_useful_block_horz_overlap, smaller_bbox, bigger_bbox = ( + check_useful_block_horizontal_overlap(useful_blocks) + ) if is_useful_block_horz_overlap: logger.warning( - f"skip this page, reason: {DropReason.USEFUL_BLOCK_HOR_OVERLAP}, smaller bbox is {smaller_bbox}, bigger bbox is {bigger_bbox}") + f'skip this page, reason: {DropReason.USEFUL_BLOCK_HOR_OVERLAP}, smaller bbox is {smaller_bbox}, bigger bbox is {bigger_bbox}' + ) # noqa: E501 for bbox in all_bboxes.copy(): if smaller_bbox == bbox[:4]: all_bboxes.remove(bbox) @@ -47,27 +56,27 @@ def remove_horizontal_overlap_block_which_smaller(all_bboxes): return is_useful_block_horz_overlap, all_bboxes -def __replace_STX_ETX(text_str:str): - """ Replace \u0002 and \u0003, as these characters become garbled when extracted using pymupdf. In fact, they were originally quotation marks. -Drawback: This issue is only observed in English text; it has not been found in Chinese text so far. +def __replace_STX_ETX(text_str: str): + """Replace \u0002 and \u0003, as these characters become garbled when extracted using pymupdf. In fact, they were originally quotation marks. + Drawback: This issue is only observed in English text; it has not been found in Chinese text so far. - Args: - text_str (str): raw text + Args: + text_str (str): raw text - Returns: - _type_: replaced text - """ + Returns: + _type_: replaced text + """ # noqa: E501 if text_str: s = text_str.replace('\u0002', "'") - s = s.replace("\u0003", "'") + s = s.replace('\u0003', "'") return s return text_str def txt_spans_extract(pdf_page, inline_equations, interline_equations): - text_raw_blocks = pdf_page.get_text("dict", flags=fitz.TEXTFLAGS_TEXT)["blocks"] - char_level_text_blocks = pdf_page.get_text("rawdict", flags=fitz.TEXTFLAGS_TEXT)[ - "blocks" + text_raw_blocks = pdf_page.get_text('dict', flags=fitz.TEXTFLAGS_TEXT)['blocks'] + char_level_text_blocks = pdf_page.get_text('rawdict', flags=fitz.TEXTFLAGS_TEXT)[ + 'blocks' ] text_blocks = combine_chars_to_pymudict(text_raw_blocks, char_level_text_blocks) text_blocks = replace_equations_in_textblock( @@ -77,54 +86,63 @@ def txt_spans_extract(pdf_page, inline_equations, interline_equations): text_blocks = remove_chars_in_text_blocks(text_blocks) spans = [] for v in text_blocks: - for line in v["lines"]: - for span in line["spans"]: - bbox = span["bbox"] + for line in v['lines']: + for span in line['spans']: + bbox = span['bbox'] if float_equal(bbox[0], bbox[2]) or float_equal(bbox[1], bbox[3]): continue - if span.get('type') not in (ContentType.InlineEquation, ContentType.InterlineEquation): + if span.get('type') not in ( + ContentType.InlineEquation, + ContentType.InterlineEquation, + ): spans.append( { - "bbox": list(span["bbox"]), - "content": __replace_STX_ETX(span["text"]), - "type": ContentType.Text, - "score": 1.0, + 'bbox': list(span['bbox']), + 'content': __replace_STX_ETX(span['text']), + 'type': ContentType.Text, + 'score': 1.0, } ) return spans def replace_text_span(pymu_spans, ocr_spans): - return list(filter(lambda x: x["type"] != ContentType.Text, ocr_spans)) + pymu_spans + return list(filter(lambda x: x['type'] != ContentType.Text, ocr_spans)) + pymu_spans def model_init(model_name: str): from transformers import LayoutLMv3ForTokenClassification + if torch.cuda.is_available(): - device = torch.device("cuda") + device = torch.device('cuda') if torch.cuda.is_bf16_supported(): supports_bfloat16 = True else: supports_bfloat16 = False else: - device = torch.device("cpu") + device = torch.device('cpu') supports_bfloat16 = False - if model_name == "layoutreader": + if model_name == 'layoutreader': # 检测modelscope的缓存目录是否存在 layoutreader_model_dir = get_local_layoutreader_model_dir() if os.path.exists(layoutreader_model_dir): - model = LayoutLMv3ForTokenClassification.from_pretrained(layoutreader_model_dir) + model = LayoutLMv3ForTokenClassification.from_pretrained( + layoutreader_model_dir + ) else: logger.warning( - f"local layoutreader model not exists, use online model from huggingface") - model = LayoutLMv3ForTokenClassification.from_pretrained("hantian/layoutreader") + 'local layoutreader model not exists, use online model from huggingface' + ) + model = LayoutLMv3ForTokenClassification.from_pretrained( + 'hantian/layoutreader' + ) # 检查设备是否支持 bfloat16 if supports_bfloat16: model.bfloat16() model.to(device).eval() else: - logger.error("model name not allow") + logger.error('model name not allow') exit(1) return model @@ -145,7 +163,9 @@ class ModelSingleton: def do_predict(boxes: List[List[int]], model) -> List[int]: - from magic_pdf.model.v3.helpers import prepare_inputs, boxes2inputs, parse_logits + from magic_pdf.model.v3.helpers import (boxes2inputs, parse_logits, + prepare_inputs) + inputs = boxes2inputs(boxes) inputs = prepare_inputs(inputs, model) logits = model(**inputs).logits.cpu().squeeze(0) @@ -154,19 +174,6 @@ def do_predict(boxes: List[List[int]], model) -> List[int]: def cal_block_index(fix_blocks, sorted_bboxes): for block in fix_blocks: - # if block['type'] in ['text', 'title', 'interline_equation']: - # line_index_list = [] - # if len(block['lines']) == 0: - # block['index'] = sorted_bboxes.index(block['bbox']) - # else: - # for line in block['lines']: - # line['index'] = sorted_bboxes.index(line['bbox']) - # line_index_list.append(line['index']) - # median_value = statistics.median(line_index_list) - # block['index'] = median_value - # - # elif block['type'] in ['table', 'image']: - # block['index'] = sorted_bboxes.index(block['bbox']) line_index_list = [] if len(block['lines']) == 0: @@ -178,9 +185,11 @@ def cal_block_index(fix_blocks, sorted_bboxes): median_value = statistics.median(line_index_list) block['index'] = median_value - # 删除图表block中的虚拟line信息 - if block['type'] in ['table', 'image']: - del block['lines'] + # 删除图表body block中的虚拟line信息, 并用real_lines信息回填 + if block['type'] in [BlockType.ImageBody, BlockType.TableBody]: + block['virtual_lines'] = copy.deepcopy(block['lines']) + block['lines'] = copy.deepcopy(block['real_lines']) + del block['real_lines'] return fix_blocks @@ -193,21 +202,22 @@ def insert_lines_into_block(block_bbox, line_height, page_w, page_h): block_weight = x1 - x0 # 如果block高度小于n行正文,则直接返回block的bbox - if line_height*3 < block_height: - if block_height > page_h*0.25 and page_w*0.5 > block_weight > page_w*0.25: # 可能是双列结构,可以切细点 - lines = int(block_height/line_height)+1 + if line_height * 3 < block_height: + if ( + block_height > page_h * 0.25 and page_w * 0.5 > block_weight > page_w * 0.25 + ): # 可能是双列结构,可以切细点 + lines = int(block_height / line_height) + 1 else: - # 如果block的宽度超过0.4页面宽度,则将block分成3行 - if block_weight > page_w*0.4: + # 如果block的宽度超过0.4页面宽度,则将block分成3行(是一种复杂布局,图不能切的太细) + if block_weight > page_w * 0.4: line_height = (y1 - y0) / 3 lines = 3 - elif block_weight > page_w*0.25: # 否则将block分成两行 - line_height = (y1 - y0) / 2 - lines = 2 - else: # 判断长宽比 - if block_height/block_weight > 1.2: # 细长的不分 + elif block_weight > page_w * 0.25: # (可能是三列结构,也切细点) + lines = int(block_height / line_height) + 1 + else: # 判断长宽比 + if block_height / block_weight > 1.2: # 细长的不分 return [[x0, y0, x1, y1]] - else: # 不细长的还是分成两行 + else: # 不细长的还是分成两行 line_height = (y1 - y0) / 2 lines = 2 @@ -229,7 +239,11 @@ def insert_lines_into_block(block_bbox, line_height, page_w, page_h): def sort_lines_by_model(fix_blocks, page_w, page_h, line_height): page_line_list = [] for block in fix_blocks: - if block['type'] in ['text', 'title', 'interline_equation']: + if block['type'] in [ + BlockType.Text, BlockType.Title, BlockType.InterlineEquation, + BlockType.ImageCaption, BlockType.ImageFootnote, + BlockType.TableCaption, BlockType.TableFootnote + ]: if len(block['lines']) == 0: bbox = block['bbox'] lines = insert_lines_into_block(bbox, line_height, page_w, page_h) @@ -240,8 +254,9 @@ def sort_lines_by_model(fix_blocks, page_w, page_h, line_height): for line in block['lines']: bbox = line['bbox'] page_line_list.append(bbox) - elif block['type'] in ['table', 'image']: + elif block['type'] in [BlockType.ImageBody, BlockType.TableBody]: bbox = block['bbox'] + block["real_lines"] = copy.deepcopy(block['lines']) lines = insert_lines_into_block(bbox, line_height, page_w, page_h) block['lines'] = [] for line in lines: @@ -256,19 +271,23 @@ def sort_lines_by_model(fix_blocks, page_w, page_h, line_height): for left, top, right, bottom in page_line_list: if left < 0: logger.warning( - f"left < 0, left: {left}, right: {right}, top: {top}, bottom: {bottom}, page_w: {page_w}, page_h: {page_h}") + f'left < 0, left: {left}, right: {right}, top: {top}, bottom: {bottom}, page_w: {page_w}, page_h: {page_h}' + ) # noqa: E501 left = 0 if right > page_w: logger.warning( - f"right > page_w, left: {left}, right: {right}, top: {top}, bottom: {bottom}, page_w: {page_w}, page_h: {page_h}") + f'right > page_w, left: {left}, right: {right}, top: {top}, bottom: {bottom}, page_w: {page_w}, page_h: {page_h}' + ) # noqa: E501 right = page_w if top < 0: logger.warning( - f"top < 0, left: {left}, right: {right}, top: {top}, bottom: {bottom}, page_w: {page_w}, page_h: {page_h}") + f'top < 0, left: {left}, right: {right}, top: {top}, bottom: {bottom}, page_w: {page_w}, page_h: {page_h}' + ) # noqa: E501 top = 0 if bottom > page_h: logger.warning( - f"bottom > page_h, left: {left}, right: {right}, top: {top}, bottom: {bottom}, page_w: {page_w}, page_h: {page_h}") + f'bottom > page_h, left: {left}, right: {right}, top: {top}, bottom: {bottom}, page_w: {page_w}, page_h: {page_h}' + ) # noqa: E501 bottom = page_h left = round(left * x_scale) @@ -276,11 +295,11 @@ def sort_lines_by_model(fix_blocks, page_w, page_h, line_height): right = round(right * x_scale) bottom = round(bottom * y_scale) assert ( - 1000 >= right >= left >= 0 and 1000 >= bottom >= top >= 0 - ), f"Invalid box. right: {right}, left: {left}, bottom: {bottom}, top: {top}" + 1000 >= right >= left >= 0 and 1000 >= bottom >= top >= 0 + ), f'Invalid box. right: {right}, left: {left}, bottom: {bottom}, top: {top}' # noqa: E126, E121 boxes.append([left, top, right, bottom]) model_manager = ModelSingleton() - model = model_manager.get_model("layoutreader") + model = model_manager.get_model('layoutreader') with torch.no_grad(): orders = do_predict(boxes, model) sorted_bboxes = [page_line_list[i] for i in orders] @@ -291,149 +310,274 @@ def sort_lines_by_model(fix_blocks, page_w, page_h, line_height): def get_line_height(blocks): page_line_height_list = [] for block in blocks: - if block['type'] in ['text', 'title', 'interline_equation']: + if block['type'] in [ + BlockType.Text, BlockType.Title, + BlockType.ImageCaption, BlockType.ImageFootnote, + BlockType.TableCaption, BlockType.TableFootnote + ]: for line in block['lines']: bbox = line['bbox'] - page_line_height_list.append(int(bbox[3]-bbox[1])) + page_line_height_list.append(int(bbox[3] - bbox[1])) if len(page_line_height_list) > 0: return statistics.median(page_line_height_list) else: return 10 -def parse_page_core(pdf_docs, magic_model, page_id, pdf_bytes_md5, imageWriter, parse_mode): +def process_groups(groups, body_key, caption_key, footnote_key): + body_blocks = [] + caption_blocks = [] + footnote_blocks = [] + for i, group in enumerate(groups): + group[body_key]['group_id'] = i + body_blocks.append(group[body_key]) + for caption_block in group[caption_key]: + caption_block['group_id'] = i + caption_blocks.append(caption_block) + for footnote_block in group[footnote_key]: + footnote_block['group_id'] = i + footnote_blocks.append(footnote_block) + return body_blocks, caption_blocks, footnote_blocks + + +def process_block_list(blocks, body_type, block_type): + indices = [block['index'] for block in blocks] + median_index = statistics.median(indices) + + body_bbox = next((block['bbox'] for block in blocks if block.get('type') == body_type), []) + + return { + 'type': block_type, + 'bbox': body_bbox, + 'blocks': blocks, + 'index': median_index, + } + + +def revert_group_blocks(blocks): + image_groups = {} + table_groups = {} + new_blocks = [] + for block in blocks: + if block['type'] in [BlockType.ImageBody, BlockType.ImageCaption, BlockType.ImageFootnote]: + group_id = block['group_id'] + if group_id not in image_groups: + image_groups[group_id] = [] + image_groups[group_id].append(block) + elif block['type'] in [BlockType.TableBody, BlockType.TableCaption, BlockType.TableFootnote]: + group_id = block['group_id'] + if group_id not in table_groups: + table_groups[group_id] = [] + table_groups[group_id].append(block) + else: + new_blocks.append(block) + + for group_id, blocks in image_groups.items(): + new_blocks.append(process_block_list(blocks, BlockType.ImageBody, BlockType.Image)) + + for group_id, blocks in table_groups.items(): + new_blocks.append(process_block_list(blocks, BlockType.TableBody, BlockType.Table)) + + return new_blocks + + +def parse_page_core( + page_doc: PageableData, magic_model, page_id, pdf_bytes_md5, imageWriter, parse_mode +): need_drop = False drop_reason = [] - '''从magic_model对象中获取后面会用到的区块信息''' - img_blocks = magic_model.get_imgs(page_id) - table_blocks = magic_model.get_tables(page_id) + """从magic_model对象中获取后面会用到的区块信息""" + # img_blocks = magic_model.get_imgs(page_id) + # table_blocks = magic_model.get_tables(page_id) + + img_groups = magic_model.get_imgs_v2(page_id) + table_groups = magic_model.get_tables_v2(page_id) + + img_body_blocks, img_caption_blocks, img_footnote_blocks = process_groups( + img_groups, 'image_body', 'image_caption_list', 'image_footnote_list' + ) + + table_body_blocks, table_caption_blocks, table_footnote_blocks = process_groups( + table_groups, 'table_body', 'table_caption_list', 'table_footnote_list' + ) + discarded_blocks = magic_model.get_discarded(page_id) text_blocks = magic_model.get_text_blocks(page_id) title_blocks = magic_model.get_title_blocks(page_id) - inline_equations, interline_equations, interline_equation_blocks = magic_model.get_equations(page_id) + inline_equations, interline_equations, interline_equation_blocks = ( + magic_model.get_equations(page_id) + ) page_w, page_h = magic_model.get_page_size(page_id) spans = magic_model.get_all_spans(page_id) - '''根据parse_mode,构造spans''' - if parse_mode == "txt": + """根据parse_mode,构造spans""" + if parse_mode == SupportedPdfParseMethod.TXT: """ocr 中文本类的 span 用 pymu spans 替换!""" - pymu_spans = txt_spans_extract( - pdf_docs[page_id], inline_equations, interline_equations - ) + pymu_spans = txt_spans_extract(page_doc, inline_equations, interline_equations) spans = replace_text_span(pymu_spans, spans) - elif parse_mode == "ocr": + elif parse_mode == SupportedPdfParseMethod.OCR: pass else: - raise Exception("parse_mode must be txt or ocr") + raise Exception('parse_mode must be txt or ocr') - '''删除重叠spans中置信度较低的那些''' + """删除重叠spans中置信度较低的那些""" spans, dropped_spans_by_confidence = remove_overlaps_low_confidence_spans(spans) - '''删除重叠spans中较小的那些''' + """删除重叠spans中较小的那些""" spans, dropped_spans_by_span_overlap = remove_overlaps_min_spans(spans) - '''对image和table截图''' - spans = ocr_cut_image_and_table(spans, pdf_docs[page_id], page_id, pdf_bytes_md5, imageWriter) + """对image和table截图""" + spans = ocr_cut_image_and_table( + spans, page_doc, page_id, pdf_bytes_md5, imageWriter + ) - '''将所有区块的bbox整理到一起''' + """将所有区块的bbox整理到一起""" # interline_equation_blocks参数不够准,后面切换到interline_equations上 interline_equation_blocks = [] if len(interline_equation_blocks) > 0: all_bboxes, all_discarded_blocks = ocr_prepare_bboxes_for_layout_split_v2( - img_blocks, table_blocks, discarded_blocks, text_blocks, title_blocks, - interline_equation_blocks, page_w, page_h) + img_body_blocks, img_caption_blocks, img_footnote_blocks, + table_body_blocks, table_caption_blocks, table_footnote_blocks, + discarded_blocks, + text_blocks, + title_blocks, + interline_equation_blocks, + page_w, + page_h, + ) else: all_bboxes, all_discarded_blocks = ocr_prepare_bboxes_for_layout_split_v2( - img_blocks, table_blocks, discarded_blocks, text_blocks, title_blocks, - interline_equations, page_w, page_h) + img_body_blocks, img_caption_blocks, img_footnote_blocks, + table_body_blocks, table_caption_blocks, table_footnote_blocks, + discarded_blocks, + text_blocks, + title_blocks, + interline_equations, + page_w, + page_h, + ) - '''先处理不需要排版的discarded_blocks''' - discarded_block_with_spans, spans = fill_spans_in_blocks(all_discarded_blocks, spans, 0.4) + """先处理不需要排版的discarded_blocks""" + discarded_block_with_spans, spans = fill_spans_in_blocks( + all_discarded_blocks, spans, 0.4 + ) fix_discarded_blocks = fix_discarded_block(discarded_block_with_spans) - '''如果当前页面没有bbox则跳过''' + """如果当前页面没有bbox则跳过""" if len(all_bboxes) == 0: - logger.warning(f"skip this page, not found useful bbox, page_id: {page_id}") - return ocr_construct_page_component_v2([], [], page_id, page_w, page_h, [], - [], [], interline_equations, fix_discarded_blocks, - need_drop, drop_reason) + logger.warning(f'skip this page, not found useful bbox, page_id: {page_id}') + return ocr_construct_page_component_v2( + [], + [], + page_id, + page_w, + page_h, + [], + [], + [], + interline_equations, + fix_discarded_blocks, + need_drop, + drop_reason, + ) - '''将span填入blocks中''' - block_with_spans, spans = fill_spans_in_blocks(all_bboxes, spans, 0.3) + """将span填入blocks中""" + block_with_spans, spans = fill_spans_in_blocks(all_bboxes, spans, 0.5) - '''对block进行fix操作''' - fix_blocks = fix_block_spans(block_with_spans, img_blocks, table_blocks) + """对block进行fix操作""" + fix_blocks = fix_block_spans_v2(block_with_spans) - '''获取所有line并计算正文line的高度''' + """获取所有line并计算正文line的高度""" line_height = get_line_height(fix_blocks) - '''获取所有line并对line排序''' + """获取所有line并对line排序""" sorted_bboxes = sort_lines_by_model(fix_blocks, page_w, page_h, line_height) - '''根据line的中位数算block的序列关系''' + """根据line的中位数算block的序列关系""" fix_blocks = cal_block_index(fix_blocks, sorted_bboxes) - '''重排block''' + """将image和table的block还原回group形式参与后续流程""" + fix_blocks = revert_group_blocks(fix_blocks) + + """重排block""" sorted_blocks = sorted(fix_blocks, key=lambda b: b['index']) - '''获取QA需要外置的list''' + """获取QA需要外置的list""" images, tables, interline_equations = get_qa_need_list_v2(sorted_blocks) - '''构造pdf_info_dict''' - page_info = ocr_construct_page_component_v2(sorted_blocks, [], page_id, page_w, page_h, [], - images, tables, interline_equations, fix_discarded_blocks, - need_drop, drop_reason) + """构造pdf_info_dict""" + page_info = ocr_construct_page_component_v2( + sorted_blocks, + [], + page_id, + page_w, + page_h, + [], + images, + tables, + interline_equations, + fix_discarded_blocks, + need_drop, + drop_reason, + ) return page_info -def pdf_parse_union(pdf_bytes, - model_list, - imageWriter, - parse_mode, - start_page_id=0, - end_page_id=None, - debug_mode=False, - ): - pdf_bytes_md5 = compute_md5(pdf_bytes) - pdf_docs = fitz.open("pdf", pdf_bytes) +def pdf_parse_union( + dataset: Dataset, + model_list, + imageWriter, + parse_mode, + start_page_id=0, + end_page_id=None, + debug_mode=False, +): + pdf_bytes_md5 = compute_md5(dataset.data_bits()) - '''初始化空的pdf_info_dict''' + """初始化空的pdf_info_dict""" pdf_info_dict = {} - '''用model_list和docs对象初始化magic_model''' - magic_model = MagicModel(model_list, pdf_docs) + """用model_list和docs对象初始化magic_model""" + magic_model = MagicModel(model_list, dataset) - '''根据输入的起始范围解析pdf''' + """根据输入的起始范围解析pdf""" # end_page_id = end_page_id if end_page_id else len(pdf_docs) - 1 - end_page_id = end_page_id if end_page_id is not None and end_page_id >= 0 else len(pdf_docs) - 1 + end_page_id = ( + end_page_id + if end_page_id is not None and end_page_id >= 0 + else len(dataset) - 1 + ) - if end_page_id > len(pdf_docs) - 1: - logger.warning("end_page_id is out of range, use pdf_docs length") - end_page_id = len(pdf_docs) - 1 + if end_page_id > len(dataset) - 1: + logger.warning('end_page_id is out of range, use pdf_docs length') + end_page_id = len(dataset) - 1 - '''初始化启动时间''' + """初始化启动时间""" start_time = time.time() - for page_id, page in enumerate(pdf_docs): - '''debug时输出每页解析的耗时''' + for page_id, page in enumerate(dataset): + """debug时输出每页解析的耗时.""" if debug_mode: time_now = time.time() logger.info( - f"page_id: {page_id}, last_page_cost_time: {get_delta_time(start_time)}" + f'page_id: {page_id}, last_page_cost_time: {get_delta_time(start_time)}' ) start_time = time_now - '''解析pdf中的每一页''' + """解析pdf中的每一页""" if start_page_id <= page_id <= end_page_id: - page_info = parse_page_core(pdf_docs, magic_model, page_id, pdf_bytes_md5, imageWriter, parse_mode) + page_info = parse_page_core( + page, magic_model, page_id, pdf_bytes_md5, imageWriter, parse_mode + ) else: - page_w = page.rect.width - page_h = page.rect.height - page_info = ocr_construct_page_component_v2([], [], page_id, page_w, page_h, [], - [], [], [], [], - True, "skip page") - pdf_info_dict[f"page_{page_id}"] = page_info + page_info = page.get_page_info() + page_w = page_info.w + page_h = page_info.h + page_info = ocr_construct_page_component_v2( + [], [], page_id, page_w, page_h, [], [], [], [], [], True, 'skip page' + ) + pdf_info_dict[f'page_{page_id}'] = page_info """分段""" para_split(pdf_info_dict, debug_mode=debug_mode) @@ -441,7 +585,7 @@ def pdf_parse_union(pdf_bytes, """dict转list""" pdf_info_list = dict_to_list(pdf_info_dict) new_pdf_info_dict = { - "pdf_info": pdf_info_list, + 'pdf_info': pdf_info_list, } clean_memory() diff --git a/magic_pdf/pipe/AbsPipe.py b/magic_pdf/pipe/AbsPipe.py index b9df6672..19841374 100644 --- a/magic_pdf/pipe/AbsPipe.py +++ b/magic_pdf/pipe/AbsPipe.py @@ -17,7 +17,7 @@ class AbsPipe(ABC): PIP_TXT = "txt" def __init__(self, pdf_bytes: bytes, model_list: list, image_writer: AbsReaderWriter, is_debug: bool = False, - start_page_id=0, end_page_id=None, lang=None): + start_page_id=0, end_page_id=None, lang=None, layout_model=None, formula_enable=None, table_enable=None): self.pdf_bytes = pdf_bytes self.model_list = model_list self.image_writer = image_writer @@ -26,6 +26,9 @@ class AbsPipe(ABC): self.start_page_id = start_page_id self.end_page_id = end_page_id self.lang = lang + self.layout_model = layout_model + self.formula_enable = formula_enable + self.table_enable = table_enable def get_compress_pdf_mid_data(self): return JsonCompressor.compress_json(self.pdf_mid_data) @@ -95,9 +98,7 @@ class AbsPipe(ABC): """ pdf_mid_data = JsonCompressor.decompress_json(compressed_pdf_mid_data) pdf_info_list = pdf_mid_data["pdf_info"] - parse_type = pdf_mid_data["_parse_type"] - lang = pdf_mid_data.get("_lang", None) - content_list = union_make(pdf_info_list, MakeMode.STANDARD_FORMAT, drop_mode, img_buket_path, parse_type, lang) + content_list = union_make(pdf_info_list, MakeMode.STANDARD_FORMAT, drop_mode, img_buket_path) return content_list @staticmethod @@ -107,9 +108,7 @@ class AbsPipe(ABC): """ pdf_mid_data = JsonCompressor.decompress_json(compressed_pdf_mid_data) pdf_info_list = pdf_mid_data["pdf_info"] - parse_type = pdf_mid_data["_parse_type"] - lang = pdf_mid_data.get("_lang", None) - md_content = union_make(pdf_info_list, md_make_mode, drop_mode, img_buket_path, parse_type, lang) + md_content = union_make(pdf_info_list, md_make_mode, drop_mode, img_buket_path) return md_content diff --git a/magic_pdf/pipe/OCRPipe.py b/magic_pdf/pipe/OCRPipe.py index 7a30776b..71002a93 100644 --- a/magic_pdf/pipe/OCRPipe.py +++ b/magic_pdf/pipe/OCRPipe.py @@ -10,8 +10,10 @@ from magic_pdf.user_api import parse_ocr_pdf class OCRPipe(AbsPipe): def __init__(self, pdf_bytes: bytes, model_list: list, image_writer: AbsReaderWriter, is_debug: bool = False, - start_page_id=0, end_page_id=None, lang=None): - super().__init__(pdf_bytes, model_list, image_writer, is_debug, start_page_id, end_page_id, lang) + start_page_id=0, end_page_id=None, lang=None, + layout_model=None, formula_enable=None, table_enable=None): + super().__init__(pdf_bytes, model_list, image_writer, is_debug, start_page_id, end_page_id, lang, + layout_model, formula_enable, table_enable) def pipe_classify(self): pass @@ -19,12 +21,14 @@ class OCRPipe(AbsPipe): def pipe_analyze(self): self.model_list = doc_analyze(self.pdf_bytes, ocr=True, start_page_id=self.start_page_id, end_page_id=self.end_page_id, - lang=self.lang) + lang=self.lang, layout_model=self.layout_model, + formula_enable=self.formula_enable, table_enable=self.table_enable) def pipe_parse(self): self.pdf_mid_data = parse_ocr_pdf(self.pdf_bytes, self.model_list, self.image_writer, is_debug=self.is_debug, start_page_id=self.start_page_id, end_page_id=self.end_page_id, - lang=self.lang) + lang=self.lang, layout_model=self.layout_model, + formula_enable=self.formula_enable, table_enable=self.table_enable) def pipe_mk_uni_format(self, img_parent_path: str, drop_mode=DropMode.WHOLE_PDF): result = super().pipe_mk_uni_format(img_parent_path, drop_mode) diff --git a/magic_pdf/pipe/TXTPipe.py b/magic_pdf/pipe/TXTPipe.py index 14c4f4e4..f0bc9b7b 100644 --- a/magic_pdf/pipe/TXTPipe.py +++ b/magic_pdf/pipe/TXTPipe.py @@ -11,8 +11,10 @@ from magic_pdf.user_api import parse_txt_pdf class TXTPipe(AbsPipe): def __init__(self, pdf_bytes: bytes, model_list: list, image_writer: AbsReaderWriter, is_debug: bool = False, - start_page_id=0, end_page_id=None, lang=None): - super().__init__(pdf_bytes, model_list, image_writer, is_debug, start_page_id, end_page_id, lang) + start_page_id=0, end_page_id=None, lang=None, + layout_model=None, formula_enable=None, table_enable=None): + super().__init__(pdf_bytes, model_list, image_writer, is_debug, start_page_id, end_page_id, lang, + layout_model, formula_enable, table_enable) def pipe_classify(self): pass @@ -20,12 +22,14 @@ class TXTPipe(AbsPipe): def pipe_analyze(self): self.model_list = doc_analyze(self.pdf_bytes, ocr=False, start_page_id=self.start_page_id, end_page_id=self.end_page_id, - lang=self.lang) + lang=self.lang, layout_model=self.layout_model, + formula_enable=self.formula_enable, table_enable=self.table_enable) def pipe_parse(self): self.pdf_mid_data = parse_txt_pdf(self.pdf_bytes, self.model_list, self.image_writer, is_debug=self.is_debug, start_page_id=self.start_page_id, end_page_id=self.end_page_id, - lang=self.lang) + lang=self.lang, layout_model=self.layout_model, + formula_enable=self.formula_enable, table_enable=self.table_enable) def pipe_mk_uni_format(self, img_parent_path: str, drop_mode=DropMode.WHOLE_PDF): result = super().pipe_mk_uni_format(img_parent_path, drop_mode) diff --git a/magic_pdf/pipe/UNIPipe.py b/magic_pdf/pipe/UNIPipe.py index 226ae48f..a1ae7f90 100644 --- a/magic_pdf/pipe/UNIPipe.py +++ b/magic_pdf/pipe/UNIPipe.py @@ -14,9 +14,11 @@ from magic_pdf.user_api import parse_union_pdf, parse_ocr_pdf class UNIPipe(AbsPipe): def __init__(self, pdf_bytes: bytes, jso_useful_key: dict, image_writer: AbsReaderWriter, is_debug: bool = False, - start_page_id=0, end_page_id=None, lang=None): + start_page_id=0, end_page_id=None, lang=None, + layout_model=None, formula_enable=None, table_enable=None): self.pdf_type = jso_useful_key["_pdf_type"] - super().__init__(pdf_bytes, jso_useful_key["model_list"], image_writer, is_debug, start_page_id, end_page_id, lang) + super().__init__(pdf_bytes, jso_useful_key["model_list"], image_writer, is_debug, start_page_id, end_page_id, + lang, layout_model, formula_enable, table_enable) if len(self.model_list) == 0: self.input_model_is_empty = True else: @@ -29,18 +31,21 @@ class UNIPipe(AbsPipe): if self.pdf_type == self.PIP_TXT: self.model_list = doc_analyze(self.pdf_bytes, ocr=False, start_page_id=self.start_page_id, end_page_id=self.end_page_id, - lang=self.lang) + lang=self.lang, layout_model=self.layout_model, + formula_enable=self.formula_enable, table_enable=self.table_enable) elif self.pdf_type == self.PIP_OCR: self.model_list = doc_analyze(self.pdf_bytes, ocr=True, start_page_id=self.start_page_id, end_page_id=self.end_page_id, - lang=self.lang) + lang=self.lang, layout_model=self.layout_model, + formula_enable=self.formula_enable, table_enable=self.table_enable) def pipe_parse(self): if self.pdf_type == self.PIP_TXT: self.pdf_mid_data = parse_union_pdf(self.pdf_bytes, self.model_list, self.image_writer, is_debug=self.is_debug, input_model_is_empty=self.input_model_is_empty, start_page_id=self.start_page_id, end_page_id=self.end_page_id, - lang=self.lang) + lang=self.lang, layout_model=self.layout_model, + formula_enable=self.formula_enable, table_enable=self.table_enable) elif self.pdf_type == self.PIP_OCR: self.pdf_mid_data = parse_ocr_pdf(self.pdf_bytes, self.model_list, self.image_writer, is_debug=self.is_debug, diff --git a/magic_pdf/pre_proc/ocr_detect_all_bboxes.py b/magic_pdf/pre_proc/ocr_detect_all_bboxes.py index 8725b884..77f242b7 100644 --- a/magic_pdf/pre_proc/ocr_detect_all_bboxes.py +++ b/magic_pdf/pre_proc/ocr_detect_all_bboxes.py @@ -1,7 +1,7 @@ from loguru import logger from magic_pdf.libs.boxbase import get_minbox_if_overlap_by_ratio, calculate_overlap_area_in_bbox1_area_ratio, \ - calculate_iou + calculate_iou, calculate_vertical_projection_overlap_ratio from magic_pdf.libs.drop_tag import DropTag from magic_pdf.libs.ocr_content_type import BlockType from magic_pdf.pre_proc.remove_bbox_overlap import remove_overlap_between_bbox_for_block @@ -60,29 +60,34 @@ def ocr_prepare_bboxes_for_layout_split(img_blocks, table_blocks, discarded_bloc return all_bboxes, all_discarded_blocks, drop_reasons -def ocr_prepare_bboxes_for_layout_split_v2(img_blocks, table_blocks, discarded_blocks, text_blocks, - title_blocks, interline_equation_blocks, page_w, page_h): +def add_bboxes(blocks, block_type, bboxes): + for block in blocks: + x0, y0, x1, y1 = block['bbox'] + if block_type in [ + BlockType.ImageBody, BlockType.ImageCaption, BlockType.ImageFootnote, + BlockType.TableBody, BlockType.TableCaption, BlockType.TableFootnote + ]: + bboxes.append([x0, y0, x1, y1, None, None, None, block_type, None, None, None, None, block["score"], block["group_id"]]) + else: + bboxes.append([x0, y0, x1, y1, None, None, None, block_type, None, None, None, None, block["score"]]) + + +def ocr_prepare_bboxes_for_layout_split_v2( + img_body_blocks, img_caption_blocks, img_footnote_blocks, + table_body_blocks, table_caption_blocks, table_footnote_blocks, + discarded_blocks, text_blocks, title_blocks, interline_equation_blocks, page_w, page_h +): all_bboxes = [] - all_discarded_blocks = [] - for image in img_blocks: - x0, y0, x1, y1 = image['bbox'] - all_bboxes.append([x0, y0, x1, y1, None, None, None, BlockType.Image, None, None, None, None, image["score"]]) - for table in table_blocks: - x0, y0, x1, y1 = table['bbox'] - all_bboxes.append([x0, y0, x1, y1, None, None, None, BlockType.Table, None, None, None, None, table["score"]]) - - for text in text_blocks: - x0, y0, x1, y1 = text['bbox'] - all_bboxes.append([x0, y0, x1, y1, None, None, None, BlockType.Text, None, None, None, None, text["score"]]) - - for title in title_blocks: - x0, y0, x1, y1 = title['bbox'] - all_bboxes.append([x0, y0, x1, y1, None, None, None, BlockType.Title, None, None, None, None, title["score"]]) - - for interline_equation in interline_equation_blocks: - x0, y0, x1, y1 = interline_equation['bbox'] - all_bboxes.append([x0, y0, x1, y1, None, None, None, BlockType.InterlineEquation, None, None, None, None, interline_equation["score"]]) + add_bboxes(img_body_blocks, BlockType.ImageBody, all_bboxes) + add_bboxes(img_caption_blocks, BlockType.ImageCaption, all_bboxes) + add_bboxes(img_footnote_blocks, BlockType.ImageFootnote, all_bboxes) + add_bboxes(table_body_blocks, BlockType.TableBody, all_bboxes) + add_bboxes(table_caption_blocks, BlockType.TableCaption, all_bboxes) + add_bboxes(table_footnote_blocks, BlockType.TableFootnote, all_bboxes) + add_bboxes(text_blocks, BlockType.Text, all_bboxes) + add_bboxes(title_blocks, BlockType.Title, all_bboxes) + add_bboxes(interline_equation_blocks, BlockType.InterlineEquation, all_bboxes) '''block嵌套问题解决''' '''文本框与标题框重叠,优先信任文本框''' @@ -96,13 +101,23 @@ def ocr_prepare_bboxes_for_layout_split_v2(img_blocks, table_blocks, discarded_b '''interline_equation框被包含在文本类型框内,且interline_equation比文本区块小很多时信任文本框,这时需要舍弃公式框''' # 通过后续大框套小框逻辑删除 - '''discarded_blocks中只保留宽度超过1/3页面宽度的,高度超过10的,处于页面下半50%区域的(限定footnote)''' + '''discarded_blocks''' + all_discarded_blocks = [] + add_bboxes(discarded_blocks, BlockType.Discarded, all_discarded_blocks) + + '''footnote识别:宽度超过1/3页面宽度的,高度超过10的,处于页面下半50%区域的''' + footnote_blocks = [] for discarded in discarded_blocks: x0, y0, x1, y1 = discarded['bbox'] - all_discarded_blocks.append([x0, y0, x1, y1, None, None, None, BlockType.Discarded, None, None, None, None, discarded["score"]]) - # 将footnote加入到all_bboxes中,用来计算layout - # if (x1 - x0) > (page_w / 3) and (y1 - y0) > 10 and y0 > (page_h / 2): - # all_bboxes.append([x0, y0, x1, y1, None, None, None, BlockType.Footnote, None, None, None, None, discarded["score"]]) + if (x1 - x0) > (page_w / 3) and (y1 - y0) > 10 and y0 > (page_h / 2): + footnote_blocks.append([x0, y0, x1, y1]) + + '''移除在footnote下面的任何框''' + need_remove_blocks = find_blocks_under_footnote(all_bboxes, footnote_blocks) + if len(need_remove_blocks) > 0: + for block in need_remove_blocks: + all_bboxes.remove(block) + all_discarded_blocks.append(block) '''经过以上处理后,还存在大框套小框的情况,则删除小框''' all_bboxes = remove_overlaps_min_blocks(all_bboxes) @@ -113,6 +128,20 @@ def ocr_prepare_bboxes_for_layout_split_v2(img_blocks, table_blocks, discarded_b return all_bboxes, all_discarded_blocks +def find_blocks_under_footnote(all_bboxes, footnote_blocks): + need_remove_blocks = [] + for block in all_bboxes: + block_x0, block_y0, block_x1, block_y1 = block[:4] + for footnote_bbox in footnote_blocks: + footnote_x0, footnote_y0, footnote_x1, footnote_y1 = footnote_bbox + # 如果footnote的纵向投影覆盖了block的纵向投影的80%且block的y0大于等于footnote的y1 + if block_y0 >= footnote_y1 and calculate_vertical_projection_overlap_ratio((block_x0, block_y0, block_x1, block_y1), footnote_bbox) >= 0.8: + if block not in need_remove_blocks: + need_remove_blocks.append(block) + break + return need_remove_blocks + + def fix_interline_equation_overlap_text_blocks_with_hi_iou(all_bboxes): # 先提取所有text和interline block text_blocks = [] diff --git a/magic_pdf/pre_proc/ocr_dict_merge.py b/magic_pdf/pre_proc/ocr_dict_merge.py index 69c4982f..1b553978 100644 --- a/magic_pdf/pre_proc/ocr_dict_merge.py +++ b/magic_pdf/pre_proc/ocr_dict_merge.py @@ -49,7 +49,7 @@ def merge_spans_to_line(spans): continue # 如果当前的span与当前行的最后一个span在y轴上重叠,则添加到当前行 - if __is_overlaps_y_exceeds_threshold(span['bbox'], current_line[-1]['bbox'], 0.6): + if __is_overlaps_y_exceeds_threshold(span['bbox'], current_line[-1]['bbox'], 0.5): current_line.append(span) else: # 否则,开始新行 @@ -153,6 +153,11 @@ def fill_spans_in_blocks(blocks, spans, radio): 'type': block_type, 'bbox': block_bbox, } + if block_type in [ + BlockType.ImageBody, BlockType.ImageCaption, BlockType.ImageFootnote, + BlockType.TableBody, BlockType.TableCaption, BlockType.TableFootnote + ]: + block_dict["group_id"] = block[-1] block_spans = [] for span in spans: span_bbox = span['bbox'] @@ -201,6 +206,27 @@ def fix_block_spans(block_with_spans, img_blocks, table_blocks): return fix_blocks +def fix_block_spans_v2(block_with_spans): + """1、img_block和table_block因为包含caption和footnote的关系,存在block的嵌套关系 + 需要将caption和footnote的text_span放入相应img_block和table_block内的 + caption_block和footnote_block中 2、同时需要删除block中的spans字段.""" + fix_blocks = [] + for block in block_with_spans: + block_type = block['type'] + + if block_type in [BlockType.Text, BlockType.Title, + BlockType.ImageCaption, BlockType.ImageFootnote, + BlockType.TableCaption, BlockType.TableFootnote + ]: + block = fix_text_block(block) + elif block_type in [BlockType.InterlineEquation, BlockType.ImageBody, BlockType.TableBody]: + block = fix_interline_block(block) + else: + continue + fix_blocks.append(block) + return fix_blocks + + def fix_discarded_block(discarded_block_with_spans): fix_discarded_blocks = [] for block in discarded_block_with_spans: diff --git a/magic_pdf/resources/model_config/model_configs.yaml b/magic_pdf/resources/model_config/model_configs.yaml index e9f0d588..e56d6ee1 100644 --- a/magic_pdf/resources/model_config/model_configs.yaml +++ b/magic_pdf/resources/model_config/model_configs.yaml @@ -1,15 +1,7 @@ -config: - device: cpu - layout: True - formula: True - table_config: - model: TableMaster - is_table_recog_enable: False - max_time: 400 - weights: - layout: Layout/model_final.pth - mfd: MFD/weights.pt - mfr: MFR/unimernet_small + layoutlmv3: Layout/LayoutLMv3/model_final.pth + doclayout_yolo: Layout/YOLO/doclayout_yolo_ft.pt + yolo_v8_mfd: MFD/YOLO/yolo_v8_ft.pt + unimernet_small: MFR/unimernet_small struct_eqtable: TabRec/StructEqTable - TableMaster: TabRec/TableMaster \ No newline at end of file + tablemaster: TabRec/TableMaster \ No newline at end of file diff --git a/magic_pdf/tools/cli.py b/magic_pdf/tools/cli.py index ef4f1f48..c19176fa 100644 --- a/magic_pdf/tools/cli.py +++ b/magic_pdf/tools/cli.py @@ -52,7 +52,7 @@ without method specified, auto will be used by default.""", help=""" Input the languages in the pdf (if known) to improve OCR accuracy. Optional. You should input "Abbreviation" with language form url: - https://paddlepaddle.github.io/PaddleOCR/en/ppocr/blog/multi_languages.html#5-support-languages-and-abbreviations + https://paddlepaddle.github.io/PaddleOCR/latest/en/ppocr/blog/multi_languages.html#5-support-languages-and-abbreviations """, default=None, ) diff --git a/magic_pdf/tools/common.py b/magic_pdf/tools/common.py index bae1224c..c98da4c3 100644 --- a/magic_pdf/tools/common.py +++ b/magic_pdf/tools/common.py @@ -6,8 +6,8 @@ import click from loguru import logger import magic_pdf.model as model_config -from magic_pdf.libs.draw_bbox import (draw_layout_bbox, draw_span_bbox, - draw_model_bbox, draw_line_sort_bbox) +from magic_pdf.libs.draw_bbox import (draw_layout_bbox, draw_line_sort_bbox, + draw_model_bbox, draw_span_bbox) from magic_pdf.libs.MakeContentConfig import DropMode, MakeMode from magic_pdf.pipe.OCRPipe import OCRPipe from magic_pdf.pipe.TXTPipe import TXTPipe @@ -46,10 +46,12 @@ def do_parse( start_page_id=0, end_page_id=None, lang=None, + layout_model=None, + formula_enable=None, + table_enable=None, ): if debug_able: logger.warning('debug mode is on') - # f_dump_content_list = True f_draw_model_bbox = True f_draw_line_sort_bbox = True @@ -64,13 +66,16 @@ def do_parse( if parse_method == 'auto': jso_useful_key = {'_pdf_type': '', 'model_list': model_list} pipe = UNIPipe(pdf_bytes, jso_useful_key, image_writer, is_debug=True, - start_page_id=start_page_id, end_page_id=end_page_id, lang=lang) + start_page_id=start_page_id, end_page_id=end_page_id, lang=lang, + layout_model=layout_model, formula_enable=formula_enable, table_enable=table_enable) elif parse_method == 'txt': pipe = TXTPipe(pdf_bytes, model_list, image_writer, is_debug=True, - start_page_id=start_page_id, end_page_id=end_page_id, lang=lang) + start_page_id=start_page_id, end_page_id=end_page_id, lang=lang, + layout_model=layout_model, formula_enable=formula_enable, table_enable=table_enable) elif parse_method == 'ocr': pipe = OCRPipe(pdf_bytes, model_list, image_writer, is_debug=True, - start_page_id=start_page_id, end_page_id=end_page_id, lang=lang) + start_page_id=start_page_id, end_page_id=end_page_id, lang=lang, + layout_model=layout_model, formula_enable=formula_enable, table_enable=table_enable) else: logger.error('unknown parse method') exit(1) diff --git a/magic_pdf/user_api.py b/magic_pdf/user_api.py index c602fc33..2a4bd59e 100644 --- a/magic_pdf/user_api.py +++ b/magic_pdf/user_api.py @@ -101,11 +101,19 @@ def parse_union_pdf(pdf_bytes: bytes, pdf_models: list, imageWriter: AbsReaderWr if pdf_info_dict is None or pdf_info_dict.get("_need_drop", False): logger.warning(f"parse_pdf_by_txt drop or error, switch to parse_pdf_by_ocr") if input_model_is_empty: - pdf_models = doc_analyze(pdf_bytes, - ocr=True, - start_page_id=start_page_id, - end_page_id=end_page_id, - lang=lang) + layout_model = kwargs.get("layout_model", None) + formula_enable = kwargs.get("formula_enable", None) + table_enable = kwargs.get("table_enable", None) + pdf_models = doc_analyze( + pdf_bytes, + ocr=True, + start_page_id=start_page_id, + end_page_id=end_page_id, + lang=lang, + layout_model=layout_model, + formula_enable=formula_enable, + table_enable=table_enable, + ) pdf_info_dict = parse_pdf(parse_pdf_by_ocr) if pdf_info_dict is None: raise Exception("Both parse_pdf_by_txt and parse_pdf_by_ocr failed.") diff --git a/magic_pdf/utils/__init__.py b/magic_pdf/utils/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/magic_pdf/utils/annotations.py b/magic_pdf/utils/annotations.py new file mode 100644 index 00000000..898d8803 --- /dev/null +++ b/magic_pdf/utils/annotations.py @@ -0,0 +1,11 @@ + +from loguru import logger + + +def ImportPIL(f): + try: + import PIL # noqa: F401 + except ImportError: + logger.error('Pillow not installed, please install by pip.') + exit(1) + return f diff --git a/docs/en/.readthedocs.yaml b/next_docs/en/.readthedocs.yaml similarity index 100% rename from docs/en/.readthedocs.yaml rename to next_docs/en/.readthedocs.yaml diff --git a/docs/en/Makefile b/next_docs/en/Makefile similarity index 100% rename from docs/en/Makefile rename to next_docs/en/Makefile diff --git a/docs/en/_static/image/logo.png b/next_docs/en/_static/image/logo.png similarity index 100% rename from docs/en/_static/image/logo.png rename to next_docs/en/_static/image/logo.png diff --git a/next_docs/en/api.rst b/next_docs/en/api.rst new file mode 100644 index 00000000..9c6a9e65 --- /dev/null +++ b/next_docs/en/api.rst @@ -0,0 +1,9 @@ +Data Api +------------------ + +.. toctree:: + :maxdepth: 2 + + api/dataset.rst + api/data_reader_writer.rst + api/read_api.rst diff --git a/next_docs/en/api/data_reader_writer.rst b/next_docs/en/api/data_reader_writer.rst new file mode 100644 index 00000000..882c974c --- /dev/null +++ b/next_docs/en/api/data_reader_writer.rst @@ -0,0 +1,44 @@ + +Data Reader Writer +-------------------- + +.. autoclass:: magic_pdf.data.data_reader_writer.DataReader + :members: + :inherited-members: + +.. autoclass:: magic_pdf.data.data_reader_writer.DataWriter + :members: + :inherited-members: + +.. autoclass:: magic_pdf.data.data_reader_writer.S3DataReader + :members: + :inherited-members: + +.. autoclass:: magic_pdf.data.data_reader_writer.S3DataWriter + :members: + :inherited-members: + +.. autoclass:: magic_pdf.data.data_reader_writer.FileBasedDataReader + :members: + :inherited-members: + +.. autoclass:: magic_pdf.data.data_reader_writer.FileBasedDataWriter + :members: + :inherited-members: + +.. autoclass:: magic_pdf.data.data_reader_writer.S3DataReader + :members: + :inherited-members: + +.. autoclass:: magic_pdf.data.data_reader_writer.S3DataWriter + :members: + :inherited-members: + +.. autoclass:: magic_pdf.data.data_reader_writer.MultiBucketS3DataReader + :members: + :inherited-members: + +.. autoclass:: magic_pdf.data.data_reader_writer.MultiBucketS3DataWriter + :members: + :inherited-members: + diff --git a/next_docs/en/api/dataset.rst b/next_docs/en/api/dataset.rst new file mode 100644 index 00000000..94a97dfe --- /dev/null +++ b/next_docs/en/api/dataset.rst @@ -0,0 +1,22 @@ +Dataset Api +------------------ + +.. autoclass:: magic_pdf.data.dataset.PageableData + :members: + :inherited-members: + +.. autoclass:: magic_pdf.data.dataset.Dataset + :members: + :inherited-members: + +.. autoclass:: magic_pdf.data.dataset.ImageDataset + :members: + :inherited-members: + +.. autoclass:: magic_pdf.data.dataset.PymuDocDataset + :members: + :inherited-members: + +.. autoclass:: magic_pdf.data.dataset.Doc + :members: + :inherited-members: diff --git a/next_docs/en/api/io.rst b/next_docs/en/api/io.rst new file mode 100644 index 00000000..e69de29b diff --git a/next_docs/en/api/read_api.rst b/next_docs/en/api/read_api.rst new file mode 100644 index 00000000..439d4c15 --- /dev/null +++ b/next_docs/en/api/read_api.rst @@ -0,0 +1,6 @@ +read_api Api +------------------ + +.. automodule:: magic_pdf.data.read_api + :members: + :inherited-members: diff --git a/next_docs/en/api/schemas.rst b/next_docs/en/api/schemas.rst new file mode 100644 index 00000000..e69de29b diff --git a/next_docs/en/api/utils.rst b/next_docs/en/api/utils.rst new file mode 100644 index 00000000..8b137891 --- /dev/null +++ b/next_docs/en/api/utils.rst @@ -0,0 +1 @@ + diff --git a/docs/en/conf.py b/next_docs/en/conf.py similarity index 100% rename from docs/en/conf.py rename to next_docs/en/conf.py diff --git a/docs/en/index.rst b/next_docs/en/index.rst similarity index 85% rename from docs/en/index.rst rename to next_docs/en/index.rst index d275dde7..61a4bc39 100644 --- a/docs/en/index.rst +++ b/next_docs/en/index.rst @@ -24,3 +24,15 @@ Welcome to the MinerU Documentation Watch Fork

+ + +API Reference +------------- + +If you are looking for information on a specific function, class or +method, this part of the documentation is for you. + +.. toctree:: + :maxdepth: 2 + + api diff --git a/docs/en/make.bat b/next_docs/en/make.bat similarity index 100% rename from docs/en/make.bat rename to next_docs/en/make.bat diff --git a/docs/requirements.txt b/next_docs/requirements.txt similarity index 51% rename from docs/requirements.txt rename to next_docs/requirements.txt index ec2c6032..ddb5027a 100644 --- a/docs/requirements.txt +++ b/next_docs/requirements.txt @@ -1,4 +1,9 @@ +boto3>=1.28.43 +loguru>=0.6.0 myst-parser +Pillow==8.4.0 +pydantic>=2.7.2,<2.8.0 +PyMuPDF>=1.24.9 sphinx sphinx-argparse sphinx-book-theme diff --git a/docs/zh_cn/.readthedocs.yaml b/next_docs/zh_cn/.readthedocs.yaml similarity index 100% rename from docs/zh_cn/.readthedocs.yaml rename to next_docs/zh_cn/.readthedocs.yaml diff --git a/docs/zh_cn/Makefile b/next_docs/zh_cn/Makefile similarity index 100% rename from docs/zh_cn/Makefile rename to next_docs/zh_cn/Makefile diff --git a/docs/zh_cn/_static/image/logo.png b/next_docs/zh_cn/_static/image/logo.png similarity index 100% rename from docs/zh_cn/_static/image/logo.png rename to next_docs/zh_cn/_static/image/logo.png diff --git a/docs/zh_cn/conf.py b/next_docs/zh_cn/conf.py similarity index 100% rename from docs/zh_cn/conf.py rename to next_docs/zh_cn/conf.py diff --git a/docs/zh_cn/index.rst b/next_docs/zh_cn/index.rst similarity index 100% rename from docs/zh_cn/index.rst rename to next_docs/zh_cn/index.rst diff --git a/docs/zh_cn/make.bat b/next_docs/zh_cn/make.bat similarity index 100% rename from docs/zh_cn/make.bat rename to next_docs/zh_cn/make.bat diff --git a/projects/README.md b/projects/README.md index cd4a7154..3eca3cec 100644 --- a/projects/README.md +++ b/projects/README.md @@ -6,5 +6,4 @@ - [gradio_app](./gradio_app/README.md): Build a web app based on gradio - [web_demo](./web_demo/README.md): MinerU online [demo](https://opendatalab.com/OpenSourceTools/Extractor/PDF/) localized deployment version - [web_api](./web_api/README.md): Web API Based on FastAPI - - +- [multi_gpu](./multi_gpu/README.md): Multi-GPU parallel processing based on LitServe diff --git a/projects/README_zh-CN.md b/projects/README_zh-CN.md index 80274512..96374cd3 100644 --- a/projects/README_zh-CN.md +++ b/projects/README_zh-CN.md @@ -6,4 +6,4 @@ - [gradio_app](./gradio_app/README_zh-CN.md): 基于 Gradio 的 Web 应用 - [web_demo](./web_demo/README_zh-CN.md): MinerU在线[demo](https://opendatalab.com/OpenSourceTools/Extractor/PDF/)本地化部署版本 - [web_api](./web_api/README.md): 基于 FastAPI 的 Web API - +- [multi_gpu](./multi_gpu/README.md): 基于 LitServe 的多 GPU 并行处理 diff --git a/projects/gradio_app/app.py b/projects/gradio_app/app.py index aa576ecb..a1d0484b 100644 --- a/projects/gradio_app/app.py +++ b/projects/gradio_app/app.py @@ -3,10 +3,12 @@ import base64 import os import time +import uuid import zipfile from pathlib import Path import re +import pymupdf from loguru import logger from magic_pdf.libs.hash_utils import compute_sha256 @@ -23,7 +25,7 @@ def read_fn(path): return disk_rw.read(os.path.basename(path), AbsReaderWriter.MODE_BIN) -def parse_pdf(doc_path, output_dir, end_page_id, is_ocr): +def parse_pdf(doc_path, output_dir, end_page_id, is_ocr, layout_mode, formula_enable, table_enable, language): os.makedirs(output_dir, exist_ok=True) try: @@ -42,6 +44,10 @@ def parse_pdf(doc_path, output_dir, end_page_id, is_ocr): parse_method, False, end_page_id=end_page_id, + layout_model=layout_mode, + formula_enable=formula_enable, + table_enable=table_enable, + lang=language, ) return local_md_dir, file_name except Exception as e: @@ -93,9 +99,10 @@ def replace_image_with_base64(markdown_text, image_dir_path): return re.sub(pattern, replace, markdown_text) -def to_markdown(file_path, end_pages, is_ocr): +def to_markdown(file_path, end_pages, is_ocr, layout_mode, formula_enable, table_enable, language): # 获取识别的md文件以及压缩包文件路径 - local_md_dir, file_name = parse_pdf(file_path, './output', end_pages - 1, is_ocr) + local_md_dir, file_name = parse_pdf(file_path, './output', end_pages - 1, is_ocr, + layout_mode, formula_enable, table_enable, language) archive_zip_path = os.path.join("./output", compute_sha256(local_md_dir) + ".zip") zip_archive_success = compress_directory_to_zip(local_md_dir, archive_zip_path) if zip_archive_success == 0: @@ -138,24 +145,71 @@ with open("header.html", "r") as file: header = file.read() +latin_lang = [ + 'af', 'az', 'bs', 'cs', 'cy', 'da', 'de', 'es', 'et', 'fr', 'ga', 'hr', + 'hu', 'id', 'is', 'it', 'ku', 'la', 'lt', 'lv', 'mi', 'ms', 'mt', 'nl', + 'no', 'oc', 'pi', 'pl', 'pt', 'ro', 'rs_latin', 'sk', 'sl', 'sq', 'sv', + 'sw', 'tl', 'tr', 'uz', 'vi', 'french', 'german' +] +arabic_lang = ['ar', 'fa', 'ug', 'ur'] +cyrillic_lang = [ + 'ru', 'rs_cyrillic', 'be', 'bg', 'uk', 'mn', 'abq', 'ady', 'kbd', 'ava', + 'dar', 'inh', 'che', 'lbe', 'lez', 'tab' +] +devanagari_lang = [ + 'hi', 'mr', 'ne', 'bh', 'mai', 'ang', 'bho', 'mah', 'sck', 'new', 'gom', + 'sa', 'bgc' +] +other_lang = ['ch', 'en', 'korean', 'japan', 'chinese_cht', 'ta', 'te', 'ka'] + +all_lang = [""] +all_lang.extend([*other_lang, *latin_lang, *arabic_lang, *cyrillic_lang, *devanagari_lang]) + + +def to_pdf(file_path): + with pymupdf.open(file_path) as f: + if f.is_pdf: + return file_path + else: + pdf_bytes = f.convert_to_pdf() + # 将pdfbytes 写入到uuid.pdf中 + # 生成唯一的文件名 + unique_filename = f"{uuid.uuid4()}.pdf" + + # 构建完整的文件路径 + tmp_file_path = os.path.join(os.path.dirname(file_path), unique_filename) + + # 将字节数据写入文件 + with open(tmp_file_path, 'wb') as tmp_pdf_file: + tmp_pdf_file.write(pdf_bytes) + + return tmp_file_path + + if __name__ == "__main__": with gr.Blocks() as demo: gr.HTML(header) with gr.Row(): with gr.Column(variant='panel', scale=5): - pdf_show = gr.Markdown() + file = gr.File(label="Please upload a PDF or image", file_types=[".pdf", ".png", ".jpeg", "jpg"]) max_pages = gr.Slider(1, 10, 5, step=1, label="Max convert pages") - with gr.Row() as bu_flow: - is_ocr = gr.Checkbox(label="Force enable OCR") + with gr.Row(): + layout_mode = gr.Dropdown(["layoutlmv3", "doclayout_yolo"], label="Layout model", value="layoutlmv3") + language = gr.Dropdown(all_lang, label="Language", value="") + with gr.Row(): + formula_enable = gr.Checkbox(label="Enable formula recognition", value=True) + is_ocr = gr.Checkbox(label="Force enable OCR", value=False) + table_enable = gr.Checkbox(label="Enable table recognition(test)", value=False) + with gr.Row(): change_bu = gr.Button("Convert") - clear_bu = gr.ClearButton([pdf_show], value="Clear") - pdf_show = PDF(label="Please upload pdf", interactive=True, height=800) + clear_bu = gr.ClearButton(value="Clear") + pdf_show = PDF(label="PDF preview", interactive=True, height=800) with gr.Accordion("Examples:"): example_root = os.path.join(os.path.dirname(__file__), "examples") gr.Examples( examples=[os.path.join(example_root, _) for _ in os.listdir(example_root) if _.endswith("pdf")], - inputs=pdf_show, + inputs=pdf_show ) with gr.Column(variant='panel', scale=5): @@ -166,7 +220,9 @@ if __name__ == "__main__": latex_delimiters=latex_delimiters, line_breaks=True) with gr.Tab("Markdown text"): md_text = gr.TextArea(lines=45, show_copy_button=True) - change_bu.click(fn=to_markdown, inputs=[pdf_show, max_pages, is_ocr], outputs=[md, md_text, output_file, pdf_show]) - clear_bu.add([md, pdf_show, md_text, output_file, is_ocr]) + file.upload(fn=to_pdf, inputs=file, outputs=pdf_show) + change_bu.click(fn=to_markdown, inputs=[pdf_show, max_pages, is_ocr, layout_mode, formula_enable, table_enable, language], + outputs=[md, md_text, output_file, pdf_show]) + clear_bu.add([file, md, pdf_show, md_text, output_file, is_ocr, table_enable, language]) - demo.launch() \ No newline at end of file + demo.launch(server_name="0.0.0.0") \ No newline at end of file diff --git a/projects/gradio_app/examples/2list_1table.pdf b/projects/gradio_app/examples/2list_1table.pdf new file mode 100644 index 00000000..dd9650bf Binary files /dev/null and b/projects/gradio_app/examples/2list_1table.pdf differ diff --git a/projects/gradio_app/examples/3list_1table.pdf b/projects/gradio_app/examples/3list_1table.pdf new file mode 100644 index 00000000..5782751a Binary files /dev/null and b/projects/gradio_app/examples/3list_1table.pdf differ diff --git a/projects/gradio_app/examples/academic_paper_formula.pdf b/projects/gradio_app/examples/academic_paper_formula.pdf old mode 100755 new mode 100644 index 9515bdfe..f1381cd2 Binary files a/projects/gradio_app/examples/academic_paper_formula.pdf and b/projects/gradio_app/examples/academic_paper_formula.pdf differ diff --git a/projects/gradio_app/examples/academic_paper_img_formula.pdf b/projects/gradio_app/examples/academic_paper_img_formula.pdf old mode 100755 new mode 100644 index 319cb873..ab8ce7ea Binary files a/projects/gradio_app/examples/academic_paper_img_formula.pdf and b/projects/gradio_app/examples/academic_paper_img_formula.pdf differ diff --git a/projects/gradio_app/examples/academic_paper_list.pdf b/projects/gradio_app/examples/academic_paper_list.pdf new file mode 100644 index 00000000..ab1d86b5 Binary files /dev/null and b/projects/gradio_app/examples/academic_paper_list.pdf differ diff --git a/projects/gradio_app/examples/complex_layout.pdf b/projects/gradio_app/examples/complex_layout.pdf new file mode 100755 index 00000000..a4fc9c0f Binary files /dev/null and b/projects/gradio_app/examples/complex_layout.pdf differ diff --git a/projects/gradio_app/examples/complex_layout_para_split_list.pdf b/projects/gradio_app/examples/complex_layout_para_split_list.pdf new file mode 100644 index 00000000..ce34c640 Binary files /dev/null and b/projects/gradio_app/examples/complex_layout_para_split_list.pdf differ diff --git a/projects/gradio_app/examples/garbled_formula.pdf b/projects/gradio_app/examples/garbled_formula.pdf old mode 100755 new mode 100644 index 5a4d5c6e..a2c11939 Binary files a/projects/gradio_app/examples/garbled_formula.pdf and b/projects/gradio_app/examples/garbled_formula.pdf differ diff --git a/projects/gradio_app/examples/garbled_formula2.pdf b/projects/gradio_app/examples/garbled_formula2.pdf deleted file mode 100755 index cec3a51b..00000000 Binary files a/projects/gradio_app/examples/garbled_formula2.pdf and /dev/null differ diff --git a/projects/gradio_app/examples/garbled_img_formula.pdf b/projects/gradio_app/examples/garbled_img_formula.pdf deleted file mode 100755 index 7f1cbf26..00000000 Binary files a/projects/gradio_app/examples/garbled_img_formula.pdf and /dev/null differ diff --git a/projects/gradio_app/examples/magazine_complex_layout_images_list.pdf b/projects/gradio_app/examples/magazine_complex_layout_images_list.pdf new file mode 100644 index 00000000..8718fc0f Binary files /dev/null and b/projects/gradio_app/examples/magazine_complex_layout_images_list.pdf differ diff --git a/projects/multi_gpu/README.md b/projects/multi_gpu/README.md new file mode 100644 index 00000000..812f8b49 --- /dev/null +++ b/projects/multi_gpu/README.md @@ -0,0 +1,46 @@ +## 项目简介 +本项目提供基于 LitServe 的多 GPU 并行处理方案。LitServe 是一个简便且灵活的 AI 模型服务引擎,基于 FastAPI 构建。它为 FastAPI 增强了批处理、流式传输和 GPU 自动扩展等功能,无需为每个模型单独重建 FastAPI 服务器。 + +## 环境配置 +请使用以下命令配置所需的环境: +```bash +pip install -U litserve python-multipart filetype +pip install -U magic-pdf[full] --extra-index-url https://wheels.myhloli.com +pip install paddlepaddle-gpu==3.0.0b1 -i https://www.paddlepaddle.org.cn/packages/stable/cu118 +``` + +## 快速使用 +### 1. 启动服务端 +以下示例展示了如何启动服务端,支持自定义设置: +```python +server = ls.LitServer( + MinerUAPI(output_dir='/tmp'), # 可自定义输出文件夹 + accelerator='cuda', # 启用 GPU 加速 + devices='auto', # "auto" 使用所有 GPU + workers_per_device=1, # 每个 GPU 启动一个服务实例 + timeout=False # 设置为 False 以禁用超时 +) +server.run(port=8000) # 设定服务端口为 8000 +``` + +启动服务端命令: +```bash +python server.py +``` + +### 2. 启动客户端 +以下代码展示了客户端的使用方式,可根据需求修改配置: +```python +files = ['demo/small_ocr.pdf'] # 替换为文件路径,支持 jpg/jpeg、png、pdf 文件 +n_jobs = np.clip(len(files), 1, 8) # 设置并发线程数,此处最大为 8,可根据自身修改 +results = Parallel(n_jobs, prefer='threads', verbose=10)( + delayed(do_parse)(p) for p in files +) +print(results) +``` + +启动客户端命令: +```bash +python client.py +``` +好了,你的文件会自动在多个 GPU 上并行处理!🍻🍻🍻 diff --git a/projects/multi_gpu/client.py b/projects/multi_gpu/client.py new file mode 100644 index 00000000..3e1c70b1 --- /dev/null +++ b/projects/multi_gpu/client.py @@ -0,0 +1,39 @@ +import base64 +import requests +import numpy as np +from loguru import logger +from joblib import Parallel, delayed + + +def to_b64(file_path): + try: + with open(file_path, 'rb') as f: + return base64.b64encode(f.read()).decode('utf-8') + except Exception as e: + raise Exception(f'File: {file_path} - Info: {e}') + + +def do_parse(file_path, url='http://127.0.0.1:8000/predict', **kwargs): + try: + response = requests.post(url, json={ + 'file': to_b64(file_path), + 'kwargs': kwargs + }) + + if response.status_code == 200: + output = response.json() + output['file_path'] = file_path + return output + else: + raise Exception(response.text) + except Exception as e: + logger.error(f'File: {file_path} - Info: {e}') + + +if __name__ == '__main__': + files = ['small_ocr.pdf'] + n_jobs = np.clip(len(files), 1, 8) + results = Parallel(n_jobs, prefer='threads', verbose=10)( + delayed(do_parse)(p) for p in files + ) + print(results) diff --git a/projects/multi_gpu/server.py b/projects/multi_gpu/server.py new file mode 100644 index 00000000..ea339a95 --- /dev/null +++ b/projects/multi_gpu/server.py @@ -0,0 +1,74 @@ +import os +import fitz +import torch +import base64 +import litserve as ls +from uuid import uuid4 +from fastapi import HTTPException +from filetype import guess_extension +from magic_pdf.tools.common import do_parse +from magic_pdf.model.doc_analyze_by_custom_model import ModelSingleton + + +class MinerUAPI(ls.LitAPI): + def __init__(self, output_dir='/tmp'): + self.output_dir = output_dir + + def setup(self, device): + if device.startswith('cuda'): + os.environ['CUDA_VISIBLE_DEVICES'] = device.split(':')[-1] + if torch.cuda.device_count() > 1: + raise RuntimeError("Remove any CUDA actions before setting 'CUDA_VISIBLE_DEVICES'.") + + model_manager = ModelSingleton() + model_manager.get_model(True, False) + model_manager.get_model(False, False) + print(f'Model initialization complete on {device}!') + + def decode_request(self, request): + file = request['file'] + file = self.to_pdf(file) + opts = request.get('kwargs', {}) + opts.setdefault('debug_able', False) + opts.setdefault('parse_method', 'auto') + return file, opts + + def predict(self, inputs): + try: + do_parse(self.output_dir, pdf_name := str(uuid4()), inputs[0], [], **inputs[1]) + return pdf_name + except Exception as e: + raise HTTPException(status_code=500, detail=str(e)) + finally: + self.clean_memory() + + def encode_response(self, response): + return {'output_dir': response} + + def clean_memory(self): + import gc + if torch.cuda.is_available(): + torch.cuda.empty_cache() + torch.cuda.ipc_collect() + gc.collect() + + def to_pdf(self, file_base64): + try: + file_bytes = base64.b64decode(file_base64) + file_ext = guess_extension(file_bytes) + with fitz.open(stream=file_bytes, filetype=file_ext) as f: + if f.is_pdf: return f.tobytes() + return f.convert_to_pdf() + except Exception as e: + raise HTTPException(status_code=500, detail=str(e)) + + +if __name__ == '__main__': + server = ls.LitServer( + MinerUAPI(output_dir='/tmp'), + accelerator='cuda', + devices='auto', + workers_per_device=1, + timeout=False + ) + server.run(port=8000) diff --git a/projects/multi_gpu/small_ocr.pdf b/projects/multi_gpu/small_ocr.pdf new file mode 100644 index 00000000..2ab92332 Binary files /dev/null and b/projects/multi_gpu/small_ocr.pdf differ diff --git a/requirements-docker.txt b/requirements-docker.txt index 74b1a70b..bf691b8d 100644 --- a/requirements-docker.txt +++ b/requirements-docker.txt @@ -5,7 +5,6 @@ PyMuPDF>=1.24.9 loguru>=0.6.0 numpy>=1.21.6,<2.0.0 fast-langdetect==0.2.0 -wordninja>=2.0.0 scikit-learn>=1.0.2 pdfminer.six==20231228 unimernet==0.2.1 @@ -15,4 +14,5 @@ paddleocr==2.7.3 paddlepaddle==3.0.0b1 pypandoc struct-eqtable==0.1.0 +doclayout-yolo==0.0.2 detectron2 diff --git a/requirements.txt b/requirements.txt index d0bd653e..eced1426 100644 --- a/requirements.txt +++ b/requirements.txt @@ -8,7 +8,6 @@ pdfminer.six==20231228 pydantic>=2.7.2,<2.8.0 PyMuPDF>=1.24.9 scikit-learn>=1.0.2 -wordninja>=2.0.0 torch>=2.2.2,<=2.3.1 transformers # The requirements.txt must ensure that only necessary external dependencies are introduced. If there are new dependencies to add, please contact the project administrator. diff --git a/setup.py b/setup.py index 51b0ba7a..0a7e8db3 100644 --- a/setup.py +++ b/setup.py @@ -45,6 +45,7 @@ if __name__ == '__main__': "paddlepaddle==2.6.1;platform_system=='Windows' or platform_system=='Darwin'", # windows版本3.0.0b1效率下降,需锁定2.6.1 "pypandoc", # 表格解析latex转html "struct-eqtable==0.1.0", # 表格解析 + "doclayout_yolo==0.0.2", # doclayout_yolo "detectron2" ], }, diff --git a/tests/test_data/__init__.py b/tests/test_data/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/test_data/assets/jsonl/test_01.jsonl b/tests/test_data/assets/jsonl/test_01.jsonl new file mode 100644 index 00000000..e3bfabb4 --- /dev/null +++ b/tests/test_data/assets/jsonl/test_01.jsonl @@ -0,0 +1 @@ +{"track_id":"e8824f5a-9fcb-4ee5-b2d4-6bf2c67019dc","path":"s3://sci-hub/enbook-scimag/78800000/libgen.scimag78872000-78872999/10.1017/cbo9780511770425.012.pdf","file_type":"pdf","content_type":"application/pdf","content_length":80078,"title":"German Idealism and the Concept of Punishment || Conclusion","remark":{"file_id":"scihub_78800000/libgen.scimag78872000-78872999.zip_10.1017/cbo9780511770425.012","file_source_type":"paper","original_file_id":"10.1017/cbo9780511770425.012","file_name":"10.1017/cbo9780511770425.012.pdf","author":"Merle, Jean-Christophe"}} diff --git a/tests/test_data/assets/jsonl/test_02.jsonl b/tests/test_data/assets/jsonl/test_02.jsonl new file mode 100644 index 00000000..cbed8675 --- /dev/null +++ b/tests/test_data/assets/jsonl/test_02.jsonl @@ -0,0 +1 @@ +{"track_id":"e8824f5a-9fcb-4ee5-b2d4-6bf2c67019dc","path":"tests/test_data/assets/pdfs/test_02.pdf","file_type":"pdf","content_type":"application/pdf","content_length":80078,"title":"German Idealism and the Concept of Punishment || Conclusion","remark":{"file_id":"scihub_78800000/libgen.scimag78872000-78872999.zip_10.1017/cbo9780511770425.012","file_source_type":"paper","original_file_id":"10.1017/cbo9780511770425.012","file_name":"10.1017/cbo9780511770425.012.pdf","author":"Merle, Jean-Christophe"}} diff --git a/tests/test_data/assets/pdfs/test_01.pdf b/tests/test_data/assets/pdfs/test_01.pdf new file mode 100644 index 00000000..229be9ce Binary files /dev/null and b/tests/test_data/assets/pdfs/test_01.pdf differ diff --git a/tests/test_data/assets/pdfs/test_02.pdf b/tests/test_data/assets/pdfs/test_02.pdf new file mode 100644 index 00000000..1adcc01c Binary files /dev/null and b/tests/test_data/assets/pdfs/test_02.pdf differ diff --git a/tests/test_data/assets/pngs/test_01.png b/tests/test_data/assets/pngs/test_01.png new file mode 100644 index 00000000..d247efc0 Binary files /dev/null and b/tests/test_data/assets/pngs/test_01.png differ diff --git a/tests/test_data/assets/pngs/test_02.png b/tests/test_data/assets/pngs/test_02.png new file mode 100644 index 00000000..1f3e7921 Binary files /dev/null and b/tests/test_data/assets/pngs/test_02.png differ diff --git a/tests/test_data/data_reader_writer/__init__.py b/tests/test_data/data_reader_writer/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/test_data/data_reader_writer/test_filebase.py b/tests/test_data/data_reader_writer/test_filebase.py new file mode 100644 index 00000000..8978db4f --- /dev/null +++ b/tests/test_data/data_reader_writer/test_filebase.py @@ -0,0 +1,24 @@ +import os +import shutil + +from magic_pdf.data.data_reader_writer import (FileBasedDataReader, + FileBasedDataWriter) + + +def test_filebased_reader_writer(): + + unitest_dir = '/tmp/magic_pdf/unittest/data/filebased_reader_writer' + sub_dir = os.path.join(unitest_dir, 'sub') + abs_fn = os.path.join(unitest_dir, 'abspath.txt') + + os.makedirs(sub_dir, exist_ok=True) + + writer = FileBasedDataWriter(sub_dir) + reader = FileBasedDataReader(sub_dir) + + writer.write('test.txt', b'hello world') + assert reader.read('test.txt') == b'hello world' + + writer.write(abs_fn, b'hello world') + assert reader.read(abs_fn) == b'hello world' + shutil.rmtree(unitest_dir) diff --git a/tests/test_data/data_reader_writer/test_multi_bucket_s3.py b/tests/test_data/data_reader_writer/test_multi_bucket_s3.py new file mode 100644 index 00000000..e032d6b8 --- /dev/null +++ b/tests/test_data/data_reader_writer/test_multi_bucket_s3.py @@ -0,0 +1,82 @@ +import json +import os + +import fitz +import pytest + +from magic_pdf.data.data_reader_writer import (MultiBucketS3DataReader, + MultiBucketS3DataWriter) +from magic_pdf.data.schemas import S3Config + + +@pytest.mark.skipif( + os.getenv('S3_ACCESS_KEY_2', None) is None, reason='need s3 config!' +) +def test_multi_bucket_s3_reader_writer(): + """test multi bucket s3 reader writer must config s3 config in the + environment export S3_BUCKET=xxx export S3_ACCESS_KEY=xxx export + S3_SECRET_KEY=xxx export S3_ENDPOINT=xxx. + + export S3_BUCKET_2=xxx export S3_ACCESS_KEY_2=xxx export S3_SECRET_KEY_2=xxx export S3_ENDPOINT_2=xxx + """ + bucket = os.getenv('S3_BUCKET', '') + ak = os.getenv('S3_ACCESS_KEY', '') + sk = os.getenv('S3_SECRET_KEY', '') + endpoint_url = os.getenv('S3_ENDPOINT', '') + + bucket_2 = os.getenv('S3_BUCKET_2', '') + ak_2 = os.getenv('S3_ACCESS_KEY_2', '') + sk_2 = os.getenv('S3_SECRET_KEY_2', '') + endpoint_url_2 = os.getenv('S3_ENDPOINT_2', '') + + s3configs = [ + S3Config( + bucket_name=bucket, access_key=ak, secret_key=sk, endpoint_url=endpoint_url + ), + S3Config( + bucket_name=bucket_2, + access_key=ak_2, + secret_key=sk_2, + endpoint_url=endpoint_url_2, + ), + ] + + reader = MultiBucketS3DataReader(default_bucket=bucket, s3_configs=s3configs) + writer = MultiBucketS3DataWriter(default_bucket=bucket, s3_configs=s3configs) + + bits = reader.read('meta-index/scihub/v001/scihub/part-66210c190659-000026.jsonl') + + assert bits == reader.read( + f's3://{bucket}/meta-index/scihub/v001/scihub/part-66210c190659-000026.jsonl' + ) + + bits = reader.read( + f's3://{bucket_2}/enbook-scimag/78800000/libgen.scimag78872000-78872999/10.1017/cbo9780511770425.012.pdf' + ) + docs = fitz.open('pdf', bits) + assert len(docs) == 10 + + bits = reader.read( + 'meta-index/scihub/v001/scihub/part-66210c190659-000026.jsonl?bytes=566,713' + ) + assert bits == reader.read_at( + 'meta-index/scihub/v001/scihub/part-66210c190659-000026.jsonl', 566, 713 + ) + assert len(json.loads(bits)) > 0 + + writer.write_string( + 'unittest/data/data_reader_writer/multi_bucket_s3_data/test01.txt', 'abc' + ) + + assert 'abc'.encode() == reader.read( + 'unittest/data/data_reader_writer/multi_bucket_s3_data/test01.txt' + ) + + writer.write( + 'unittest/data/data_reader_writer/multi_bucket_s3_data/test02.txt', + '123'.encode(), + ) + + assert '123'.encode() == reader.read( + 'unittest/data/data_reader_writer/multi_bucket_s3_data/test02.txt' + ) diff --git a/tests/test_data/data_reader_writer/test_s3.py b/tests/test_data/data_reader_writer/test_s3.py new file mode 100644 index 00000000..aaa9da12 --- /dev/null +++ b/tests/test_data/data_reader_writer/test_s3.py @@ -0,0 +1,53 @@ +import json +import os + +import pytest + +from magic_pdf.data.data_reader_writer import S3DataReader, S3DataWriter + + +@pytest.mark.skipif( + os.getenv('S3_ACCESS_KEY', None) is None, reason='need s3 config!' +) +def test_multi_bucket_s3_reader_writer(): + """test multi bucket s3 reader writer must config s3 config in the + environment export S3_BUCKET=xxx export S3_ACCESS_KEY=xxx export + S3_SECRET_KEY=xxx export S3_ENDPOINT=xxx.""" + bucket = os.getenv('S3_BUCKET', '') + ak = os.getenv('S3_ACCESS_KEY', '') + sk = os.getenv('S3_SECRET_KEY', '') + endpoint_url = os.getenv('S3_ENDPOINT', '') + + reader = S3DataReader(bucket=bucket, ak=ak, sk=sk, endpoint_url=endpoint_url) + writer = S3DataWriter(bucket=bucket, ak=ak, sk=sk, endpoint_url=endpoint_url) + + bits = reader.read('meta-index/scihub/v001/scihub/part-66210c190659-000026.jsonl') + + assert bits == reader.read( + f's3://{bucket}/meta-index/scihub/v001/scihub/part-66210c190659-000026.jsonl' + ) + + bits = reader.read( + 'meta-index/scihub/v001/scihub/part-66210c190659-000026.jsonl?bytes=566,713' + ) + assert bits == reader.read_at( + 'meta-index/scihub/v001/scihub/part-66210c190659-000026.jsonl', 566, 713 + ) + assert len(json.loads(bits)) > 0 + + writer.write_string( + 'unittest/data/data_reader_writer/multi_bucket_s3_data/test01.txt', 'abc' + ) + + assert 'abc'.encode() == reader.read( + 'unittest/data/data_reader_writer/multi_bucket_s3_data/test01.txt' + ) + + writer.write( + f'{bucket}/unittest/data/data_reader_writer/multi_bucket_s3_data/test02.txt', + '123'.encode(), + ) + + assert '123'.encode() == reader.read( + 'unittest/data/data_reader_writer/multi_bucket_s3_data/test02.txt' + ) diff --git a/tests/test_data/io/__init__.py b/tests/test_data/io/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/test_data/io/test_s3.py b/tests/test_data/io/test_s3.py new file mode 100644 index 00000000..ce84a2a8 --- /dev/null +++ b/tests/test_data/io/test_s3.py @@ -0,0 +1,55 @@ +import json +import os + +import pytest + +from magic_pdf.data.io.s3 import S3Reader, S3Writer + + +@pytest.mark.skipif( + os.getenv('S3_ACCESS_KEY', None) is None, reason='s3 config not found' +) +def test_s3_reader(): + """test s3 reader. + + must config s3 config in the environment export S3_BUCKET=xxx export S3_ACCESS_KEY=xxx export S3_SECRET_KEY=xxx + export S3_ENDPOINT=xxx + """ + + bucket = os.getenv('S3_BUCKET', '') + ak = os.getenv('S3_ACCESS_KEY', '') + sk = os.getenv('S3_SECRET_KEY', '') + endpoint_url = os.getenv('S3_ENDPOINT', '') + reader = S3Reader(bucket=bucket, ak=ak, sk=sk, endpoint_url=endpoint_url) + bits = reader.read( + 'meta-index/scihub/v001/scihub/part-66210c190659-000026.jsonl' + ) + assert len(bits) > 0 + + bits = reader.read_at( + 'meta-index/scihub/v001/scihub/part-66210c190659-000026.jsonl', + 566, + 713, + ) + assert len(json.loads(bits)) > 0 + + +@pytest.mark.skipif( + os.getenv('S3_ACCESS_KEY', None) is None, reason='s3 config not found' +) +def test_s3_writer(): + """test s3 reader. + + must config s3 config in the environment export S3_BUCKET=xxx export S3_ACCESS_KEY=xxx export S3_SECRET_KEY=xxx + export S3_ENDPOINT=xxx + """ + bucket = os.getenv('S3_BUCKET', '') + ak = os.getenv('S3_ACCESS_KEY', '') + sk = os.getenv('S3_SECRET_KEY', '') + endpoint_url = os.getenv('S3_ENDPOINT', '') + writer = S3Writer(bucket=bucket, ak=ak, sk=sk, endpoint_url=endpoint_url) + test_fn = 'unittest/io/test.jsonl' + writer.write(test_fn, '123'.encode()) + reader = S3Reader(bucket=bucket, ak=ak, sk=sk, endpoint_url=endpoint_url) + bits = reader.read(test_fn) + assert bits.decode() == '123' diff --git a/tests/test_data/test_dataset.py b/tests/test_data/test_dataset.py new file mode 100644 index 00000000..8e3b186c --- /dev/null +++ b/tests/test_data/test_dataset.py @@ -0,0 +1,18 @@ + +from magic_pdf.data.dataset import ImageDataset, PymuDocDataset + + +def test_pymudataset(): + with open('tests/test_data/assets/pdfs/test_01.pdf', 'rb') as f: + bits = f.read() + datasets = PymuDocDataset(bits) + assert len(datasets) > 0 + assert datasets.get_page(0).get_page_info().h > 100 + + +def test_imagedataset(): + with open('tests/test_data/assets/pngs/test_01.png', 'rb') as f: + bits = f.read() + datasets = ImageDataset(bits) + assert len(datasets) == 1 + assert datasets.get_page(0).get_page_info().w > 100 diff --git a/tests/test_data/test_read_api.py b/tests/test_data/test_read_api.py new file mode 100644 index 00000000..a8fcf356 --- /dev/null +++ b/tests/test_data/test_read_api.py @@ -0,0 +1,78 @@ +import os + +import pytest + +from magic_pdf.data.data_reader_writer import MultiBucketS3DataReader +from magic_pdf.data.read_api import (read_jsonl, read_local_images, + read_local_pdfs) +from magic_pdf.data.schemas import S3Config + + +def test_read_local_pdfs(): + datasets = read_local_pdfs('tests/test_data/assets/pdfs') + assert len(datasets) == 2 + assert len(datasets[0]) > 0 + assert len(datasets[1]) > 0 + + assert datasets[0].get_page(0).get_page_info().w > 0 + assert datasets[0].get_page(0).get_page_info().h > 0 + + +def test_read_local_images(): + datasets = read_local_images('tests/test_data/assets/pngs', suffixes=['png']) + assert len(datasets) == 2 + assert len(datasets[0]) == 1 + assert len(datasets[1]) == 1 + + assert datasets[0].get_page(0).get_page_info().w > 0 + assert datasets[0].get_page(0).get_page_info().h > 0 + + +@pytest.mark.skipif( + os.getenv('S3_ACCESS_KEY_2', None) is None, reason='need s3 config!' +) +def test_read_json(): + """test multi bucket s3 reader writer must config s3 config in the + environment export S3_BUCKET=xxx export S3_ACCESS_KEY=xxx export + S3_SECRET_KEY=xxx export S3_ENDPOINT=xxx. + + export S3_BUCKET_2=xxx export S3_ACCESS_KEY_2=xxx export S3_SECRET_KEY_2=xxx export S3_ENDPOINT_2=xxx + """ + bucket = os.getenv('S3_BUCKET', '') + ak = os.getenv('S3_ACCESS_KEY', '') + sk = os.getenv('S3_SECRET_KEY', '') + endpoint_url = os.getenv('S3_ENDPOINT', '') + + bucket_2 = os.getenv('S3_BUCKET_2', '') + ak_2 = os.getenv('S3_ACCESS_KEY_2', '') + sk_2 = os.getenv('S3_SECRET_KEY_2', '') + endpoint_url_2 = os.getenv('S3_ENDPOINT_2', '') + + s3configs = [ + S3Config( + bucket_name=bucket, access_key=ak, secret_key=sk, endpoint_url=endpoint_url + ), + S3Config( + bucket_name=bucket_2, + access_key=ak_2, + secret_key=sk_2, + endpoint_url=endpoint_url_2, + ), + ] + + reader = MultiBucketS3DataReader(bucket, s3configs) + + datasets = read_jsonl( + f's3://{bucket}/meta-index/scihub/v001/scihub/part-66210c190659-000026.jsonl', + reader, + ) + assert len(datasets) > 0 + assert len(datasets[0]) == 10 + + datasets = read_jsonl('tests/test_data/assets/jsonl/test_01.jsonl', reader) + assert len(datasets) == 1 + assert len(datasets[0]) == 10 + + datasets = read_jsonl('tests/test_data/assets/jsonl/test_02.jsonl') + assert len(datasets) == 1 + assert len(datasets[0]) == 1 diff --git a/tests/test_model/__init__.py b/tests/test_model/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/test_model/assets/test_01.model.json b/tests/test_model/assets/test_01.model.json new file mode 100644 index 00000000..ee79509e --- /dev/null +++ b/tests/test_model/assets/test_01.model.json @@ -0,0 +1,687 @@ +[ + { + "layout_dets": [ + { + "category_id": 3, + "poly": [ + 776.7277221679688, + 688.448974609375, + 1242.224365234375, + 688.448974609375, + 1242.224365234375, + 1182.0628662109375, + 776.7277221679688, + 1182.0628662109375 + ], + "score": 0.999997079372406 + }, + { + "category_id": 3, + "poly": [ + 775.9269409179688, + 1389.754638671875, + 1243.672119140625, + 1389.754638671875, + 1243.672119140625, + 1859.716064453125, + 775.9269409179688, + 1859.716064453125 + ], + "score": 0.9999949932098389 + }, + { + "category_id": 1, + "poly": [ + 752.11572265625, + 1939.3634033203125, + 1430.1146240234375, + 1939.3634033203125, + 1430.1146240234375, + 2041.1771240234375, + 752.11572265625, + 2041.1771240234375 + ], + "score": 0.999975323677063 + }, + { + "category_id": 3, + "poly": [ + 46.55152893066406, + 686.12939453125, + 638.8861083984375, + 686.12939453125, + 638.8861083984375, + 1803.419189453125, + 46.55152893066406, + 1803.419189453125 + ], + "score": 0.999961256980896 + }, + { + "category_id": 3, + "poly": [ + 33.684722900390625, + 150.77980041503906, + 1238.0679931640625, + 150.77980041503906, + 1238.0679931640625, + 524.98291015625, + 33.684722900390625, + 524.98291015625 + ], + "score": 0.9999504089355469 + }, + { + "category_id": 1, + "poly": [ + 24.685693740844727, + 1875.9998779296875, + 703.5064697265625, + 1875.9998779296875, + 703.5064697265625, + 2050.7431640625, + 24.685693740844727, + 2050.7431640625 + ], + "score": 0.9999105334281921 + }, + { + "category_id": 1, + "poly": [ + 750.97705078125, + 1252.206787109375, + 1430.0809326171875, + 1252.206787109375, + 1430.0809326171875, + 1357.2947998046875, + 750.97705078125, + 1357.2947998046875 + ], + "score": 0.999853789806366 + }, + { + "category_id": 4, + "poly": [ + 904.842041015625, + 1213.027099609375, + 1273.5655517578125, + 1213.027099609375, + 1273.5655517578125, + 1242.717529296875, + 904.842041015625, + 1242.717529296875 + ], + "score": 0.9995817542076111 + }, + { + "category_id": 4, + "poly": [ + 905.3208618164062, + 1898.5325927734375, + 1273.1282958984375, + 1898.5325927734375, + 1273.1282958984375, + 1928.9906005859375, + 905.3208618164062, + 1928.9906005859375 + ], + "score": 0.9986443519592285 + }, + { + "category_id": 4, + "poly": [ + 372.0135498046875, + 556.02685546875, + 1084.9647216796875, + 556.02685546875, + 1084.9647216796875, + 586.6792602539062, + 372.0135498046875, + 586.6792602539062 + ], + "score": 0.9985352754592896 + }, + { + "category_id": 2, + "poly": [ + 1350.63671875, + 79.77919006347656, + 1379.6220703125, + 79.77919006347656, + 1379.6220703125, + 99.83788299560547, + 1350.63671875, + 99.83788299560547 + ], + "score": 0.9973036646842957 + }, + { + "category_id": 4, + "poly": [ + 203.2659912109375, + 597.2034912109375, + 1251.0240478515625, + 597.2034912109375, + 1251.0240478515625, + 657.985595703125, + 203.2659912109375, + 657.985595703125 + ], + "score": 0.9622809886932373 + }, + { + "category_id": 0, + "poly": [ + 70.87332916259766, + 1834.5714111328125, + 657.8504638671875, + 1834.5714111328125, + 657.8504638671875, + 1865.07373046875, + 70.87332916259766, + 1865.07373046875 + ], + "score": 0.8580453395843506 + }, + { + "category_id": 1, + "poly": [ + 189.0360870361328, + 597.2406616210938, + 1252.3204345703125, + 597.2406616210938, + 1252.3204345703125, + 658.4781494140625, + 189.0360870361328, + 658.4781494140625 + ], + "score": 0.3083903193473816 + }, + { + "category_id": 13, + "poly": [ + 1190, + 1980, + 1206, + 1980, + 1206, + 1997, + 1190, + 1997 + ], + "score": 0.51, + "latex": ":" + }, + { + "category_id": 13, + "poly": [ + 1219, + 1331, + 1235, + 1331, + 1235, + 1348, + 1219, + 1348 + ], + "score": 0.49, + "latex": ":" + }, + { + "category_id": 13, + "poly": [ + 798, + 2016, + 813, + 2016, + 813, + 2033, + 798, + 2033 + ], + "score": 0.41, + "latex": ":" + }, + { + "category_id": 13, + "poly": [ + 135, + 1991, + 148, + 1991, + 148, + 2006, + 135, + 2006 + ], + "score": 0.39, + "latex": ":" + }, + { + "category_id": 13, + "poly": [ + 400, + 1916, + 416, + 1916, + 416, + 1933, + 400, + 1933 + ], + "score": 0.38, + "latex": ":" + }, + { + "category_id": 13, + "poly": [ + 1148, + 1944, + 1162, + 1944, + 1162, + 1961, + 1148, + 1961 + ], + "score": 0.31, + "latex": ":" + }, + { + "category_id": 15, + "poly": [ + 798.0, + 1943.0, + 1147.0, + 1943.0, + 1147.0, + 1968.0, + 798.0, + 1968.0 + ], + "score": 0.95, + "text": "Fig 4 SSCP analysis of FHIT exon 4. T" + }, + { + "category_id": 15, + "poly": [ + 1163.0, + 1943.0, + 1425.0, + 1943.0, + 1425.0, + 1968.0, + 1163.0, + 1968.0 + ], + "score": 0.96, + "text": "Tumor tissue ; N :Corresponding" + }, + { + "category_id": 15, + "poly": [ + 755.0, + 1979.0, + 1189.0, + 1979.0, + 1189.0, + 2004.0, + 755.0, + 2004.0 + ], + "score": 0.92, + "text": "normal tissue ; M : PBR322/Hae II Marker ; ssDNA" + }, + { + "category_id": 15, + "poly": [ + 1207.0, + 1979.0, + 1422.0, + 1979.0, + 1422.0, + 2004.0, + 1207.0, + 2004.0 + ], + "score": 0.97, + "text": "Single-stranded DNA ; ds-" + }, + { + "category_id": 15, + "poly": [ + 755.0, + 2015.0, + 797.0, + 2015.0, + 797.0, + 2038.0, + 755.0, + 2038.0 + ], + "score": 1.0, + "text": "DNA" + }, + { + "category_id": 15, + "poly": [ + 814.0, + 2015.0, + 996.0, + 2015.0, + 996.0, + 2038.0, + 814.0, + 2038.0 + ], + "score": 0.98, + "text": "Double-stranded DNA" + }, + { + "category_id": 15, + "poly": [ + 71.0, + 1880.0, + 698.0, + 1880.0, + 698.0, + 1902.0, + 71.0, + 1902.0 + ], + "score": 0.96, + "text": "Fig 2Alterations of PCR amplified products of FHIT exon 3,4,5 and" + }, + { + "category_id": 15, + "poly": [ + 28.0, + 1916.0, + 399.0, + 1916.0, + 399.0, + 1937.0, + 28.0, + 1937.0 + ], + "score": 0.98, + "text": "microsatellite marker D3S1300、D3S1312.A" + }, + { + "category_id": 15, + "poly": [ + 417.0, + 1916.0, + 701.0, + 1916.0, + 701.0, + 1937.0, + 417.0, + 1937.0 + ], + "score": 0.9, + "text": "Deletion of exon5(arrows);B :" + }, + { + "category_id": 15, + "poly": [ + 29.0, + 1953.0, + 700.0, + 1953.0, + 700.0, + 1974.0, + 29.0, + 1974.0 + ], + "score": 0.95, + "text": "Deletion of exon 3 A( arrows);C : Deletion of microsatellite marker D3S1300," + }, + { + "category_id": 15, + "poly": [ + 28.0, + 1989.0, + 134.0, + 1989.0, + 134.0, + 2014.0, + 28.0, + 2014.0 + ], + "score": 1.0, + "text": "D3S1312.T" + }, + { + "category_id": 15, + "poly": [ + 149.0, + 1989.0, + 696.0, + 1989.0, + 696.0, + 2014.0, + 149.0, + 2014.0 + ], + "score": 0.96, + "text": "Tumor ; N : Corresponding normal tissue ; L : Corresponding lymph" + }, + { + "category_id": 15, + "poly": [ + 30.0, + 2027.0, + 634.0, + 2027.0, + 634.0, + 2047.0, + 30.0, + 2047.0 + ], + "score": 0.94, + "text": "node tissue;M :DL2000 DNA marker;L1:Lewis ;A :A549;S SPAC-1" + }, + { + "category_id": 15, + "poly": [ + 801.0, + 1259.0, + 1427.0, + 1259.0, + 1427.0, + 1280.0, + 801.0, + 1280.0 + ], + "score": 0.94, + "text": "Fig 3SSCP analysis of FHIT exon 3.The arrow indicateda deletion of" + }, + { + "category_id": 15, + "poly": [ + 757.0, + 1294.0, + 1424.0, + 1294.0, + 1424.0, + 1318.0, + 757.0, + 1318.0 + ], + "score": 0.96, + "text": "exon 3 of 41T. T : Tumor tissue ; N : Corresponding normal tissue ; M PBR322/" + }, + { + "category_id": 15, + "poly": [ + 755.0, + 1329.0, + 1218.0, + 1329.0, + 1218.0, + 1355.0, + 755.0, + 1355.0 + ], + "score": 0.95, + "text": "Hae Il Marker / ssDNA : Single-stranded DNA ; dsDNA" + }, + { + "category_id": 15, + "poly": [ + 1236.0, + 1329.0, + 1418.0, + 1329.0, + 1418.0, + 1355.0, + 1236.0, + 1355.0 + ], + "score": 1.0, + "text": "Double-strandedDNA" + }, + { + "category_id": 15, + "poly": [ + 910.0, + 1217.0, + 1269.0, + 1217.0, + 1269.0, + 1241.0, + 910.0, + 1241.0 + ], + "score": 1.0, + "text": "图3FHIT基因外显子3的SSCP分析" + }, + { + "category_id": 15, + "poly": [ + 909.0, + 1904.0, + 1269.0, + 1904.0, + 1269.0, + 1927.0, + 909.0, + 1927.0 + ], + "score": 1.0, + "text": "图4FHIT基因外显子4的SSCP分析" + }, + { + "category_id": 15, + "poly": [ + 374.0, + 563.0, + 1077.0, + 563.0, + 1077.0, + 583.0, + 374.0, + 583.0 + ], + "score": 0.99, + "text": "图1FHIT基因外显子3、4、5、8和微卫星灶的PCR扩增产物琼脂糖电泳图" + }, + { + "category_id": 15, + "poly": [ + 1351.0, + 81.0, + 1376.0, + 81.0, + 1376.0, + 102.0, + 1351.0, + 102.0 + ], + "score": 1.0, + "text": "13" + }, + { + "category_id": 15, + "poly": [ + 207.0, + 600.0, + 1245.0, + 600.0, + 1245.0, + 624.0, + 207.0, + 624.0 + ], + "score": 0.96, + "text": "Fig 1 Agarose electrophoresis of PCR products of exor( A)3 ,4 ,5 ,8 and three microsatellite markers( B)of FHIT gene" + }, + { + "category_id": 15, + "poly": [ + 309.0, + 634.0, + 1142.0, + 634.0, + 1142.0, + 662.0, + 309.0, + 662.0 + ], + "score": 0.97, + "text": "M1 :DL2000 DNA marker ; M2 PBR322/Hae Il marker ; T :Tumor ; N :Corresponding normal tissue" + }, + { + "category_id": 15, + "poly": [ + 73.0, + 1840.0, + 651.0, + 1840.0, + 651.0, + 1864.0, + 73.0, + 1864.0 + ], + "score": 1.0, + "text": "图2FHIT基因外显子和微卫星灶PCR扩增产物缺失电泳图" + }, + { + "category_id": 15, + "poly": [ + 207.0, + 600.0, + 1245.0, + 600.0, + 1245.0, + 625.0, + 207.0, + 625.0 + ], + "score": 0.96, + "text": "Fig 1 Agarose electrophoresis of PCR products of exor A)3 ,4 ,5 ,8 and three microsatellite markers( B)of FHIT gene" + }, + { + "category_id": 15, + "poly": [ + 309.0, + 635.0, + 1142.0, + 635.0, + 1142.0, + 661.0, + 309.0, + 661.0 + ], + "score": 0.97, + "text": "M1 :DL2000 DNA marker ; M2 PBR322/Hae Il marker ; T Tumor ; N :Corresponding normal tissue" + } + ], + "page_info": { + "page_no": 0, + "height": 2080, + "width": 1472 + } + } +] diff --git a/tests/test_model/assets/test_01.pdf b/tests/test_model/assets/test_01.pdf new file mode 100644 index 00000000..e4500499 Binary files /dev/null and b/tests/test_model/assets/test_01.pdf differ diff --git a/tests/test_model/assets/test_02.model.json b/tests/test_model/assets/test_02.model.json new file mode 100644 index 00000000..0a308987 --- /dev/null +++ b/tests/test_model/assets/test_02.model.json @@ -0,0 +1,17564 @@ +[ + { + "layout_dets": [ + { + "category_id": 2, + "poly": [ + 118.60955810546875, + 198.658203125, + 267.46044921875, + 198.658203125, + 267.46044921875, + 363.13531494140625, + 118.60955810546875, + 363.13531494140625 + ], + "score": 0.9999977946281433 + }, + { + "category_id": 2, + "poly": [ + 1082.397216796875, + 196.80734252929688, + 1380.781005859375, + 196.80734252929688, + 1380.781005859375, + 394.29400634765625, + 1082.397216796875, + 394.29400634765625 + ], + "score": 0.9999669790267944 + }, + { + "category_id": 2, + "poly": [ + 117.83770751953125, + 1687.9595947265625, + 1381.0810546875, + 1687.9595947265625, + 1381.0810546875, + 1765.1331787109375, + 117.83770751953125, + 1765.1331787109375 + ], + "score": 0.9999470114707947 + }, + { + "category_id": 1, + "poly": [ + 212.48126220703125, + 622.498291015625, + 1290.409423828125, + 622.498291015625, + 1290.409423828125, + 731.6904296875, + 212.48126220703125, + 731.6904296875 + ], + "score": 0.9999340772628784 + }, + { + "category_id": 0, + "poly": [ + 244.640625, + 473.26220703125, + 1256.727294921875, + 473.26220703125, + 1256.727294921875, + 519.3681640625, + 244.640625, + 519.3681640625 + ], + "score": 0.9999324083328247 + }, + { + "category_id": 1, + "poly": [ + 391.2038269042969, + 752.9738159179688, + 1106.6009521484375, + 752.9738159179688, + 1106.6009521484375, + 773.8135986328125, + 391.2038269042969, + 773.8135986328125 + ], + "score": 0.999659538269043 + }, + { + "category_id": 1, + "poly": [ + 116.69463348388672, + 912.680908203125, + 1383.009521484375, + 912.680908203125, + 1383.009521484375, + 1526.5164794921875, + 116.69463348388672, + 1526.5164794921875 + ], + "score": 0.9996497631072998 + }, + { + "category_id": 2, + "poly": [ + 556.8428344726562, + 344.6543273925781, + 942.172119140625, + 344.6543273925781, + 942.172119140625, + 368.55316162109375, + 556.8428344726562, + 368.55316162109375 + ], + "score": 0.9996120929718018 + }, + { + "category_id": 0, + "poly": [ + 118.258544921875, + 864.1715087890625, + 210.07864379882812, + 864.1715087890625, + 210.07864379882812, + 889.3430786132812, + 118.258544921875, + 889.3430786132812 + ], + "score": 0.999344527721405 + }, + { + "category_id": 1, + "poly": [ + 241.03976440429688, + 551.4166870117188, + 1255.7645263671875, + 551.4166870117188, + 1255.7645263671875, + 595.4854736328125, + 241.03976440429688, + 595.4854736328125 + ], + "score": 0.9993418455123901 + }, + { + "category_id": 2, + "poly": [ + 117.89942169189453, + 1794.3287353515625, + 772.7922973632812, + 1794.3287353515625, + 772.7922973632812, + 1842.42919921875, + 117.89942169189453, + 1842.42919921875 + ], + "score": 0.9991039633750916 + }, + { + "category_id": 2, + "poly": [ + 515.6521606445312, + 193.61793518066406, + 985.9738159179688, + 193.61793518066406, + 985.9738159179688, + 291.953125, + 515.6521606445312, + 291.953125 + ], + "score": 0.9970645904541016 + }, + { + "category_id": 1, + "poly": [ + 117.53016662597656, + 1570.8095703125, + 865.2678833007812, + 1570.8095703125, + 865.2678833007812, + 1593.182861328125, + 117.53016662597656, + 1593.182861328125 + ], + "score": 0.9883127212524414 + }, + { + "category_id": 1, + "poly": [ + 119.48209381103516, + 1508.9144287109375, + 539.9886474609375, + 1508.9144287109375, + 539.9886474609375, + 1534.0999755859375, + 119.48209381103516, + 1534.0999755859375 + ], + "score": 0.8136677742004395 + }, + { + "category_id": 2, + "poly": [ + 1083.8271484375, + 374.8357238769531, + 1380.78369140625, + 374.8357238769531, + 1380.78369140625, + 395.9932861328125, + 1083.8271484375, + 395.9932861328125 + ], + "score": 0.3611733317375183 + }, + { + "category_id": 2, + "poly": [ + 515.6149291992188, + 196.63461303710938, + 984.0328369140625, + 196.63461303710938, + 984.0328369140625, + 221.70213317871094, + 515.6149291992188, + 221.70213317871094 + ], + "score": 0.33345672488212585 + }, + { + "category_id": 13, + "poly": [ + 714, + 1383, + 767, + 1383, + 767, + 1411, + 714, + 1411 + ], + "score": 0.89, + "latex": "N_{\\mathrm{zero}}" + }, + { + "category_id": 13, + "poly": [ + 571, + 1351, + 636, + 1351, + 636, + 1380, + 571, + 1380 + ], + "score": 0.87, + "latex": "(N_{\\mathrm{zero}})" + }, + { + "category_id": 13, + "poly": [ + 398, + 1793, + 419, + 1793, + 419, + 1815, + 398, + 1815 + ], + "score": 0.75, + "latex": "\\circledcirc" + }, + { + "category_id": 13, + "poly": [ + 116, + 1509, + 140, + 1509, + 140, + 1533, + 116, + 1533 + ], + "score": 0.73, + "latex": "\\copyright" + }, + { + "category_id": 13, + "poly": [ + 315, + 1713, + 479, + 1713, + 479, + 1739, + 315, + 1739 + ], + "score": 0.36, + "latex": "+61\\ 3\\ 9450\\ 8719" + }, + { + "category_id": 13, + "poly": [ + 148, + 1743, + 166, + 1743, + 166, + 1765, + 148, + 1765 + ], + "score": 0.35, + "latex": "E" + }, + { + "category_id": 13, + "poly": [ + 369, + 1743, + 387, + 1743, + 387, + 1764, + 369, + 1764 + ], + "score": 0.26, + "latex": "@" + }, + { + "category_id": 15, + "poly": [ + 124.0, + 343.0, + 263.0, + 343.0, + 263.0, + 364.0, + 124.0, + 364.0 + ], + "score": 0.93, + "text": "ELSEVIER" + }, + { + "category_id": 15, + "poly": [ + 1168.0, + 218.0, + 1284.0, + 218.0, + 1284.0, + 251.0, + 1168.0, + 251.0 + ], + "score": 0.99, + "text": "Journal" + }, + { + "category_id": 15, + "poly": [ + 1173.0, + 259.0, + 1206.0, + 259.0, + 1206.0, + 287.0, + 1173.0, + 287.0 + ], + "score": 0.93, + "text": "Of" + }, + { + "category_id": 15, + "poly": [ + 1163.0, + 296.0, + 1376.0, + 296.0, + 1376.0, + 346.0, + 1163.0, + 346.0 + ], + "score": 1.0, + "text": "Hydrology" + }, + { + "category_id": 15, + "poly": [ + 1084.0, + 376.0, + 1379.0, + 376.0, + 1379.0, + 393.0, + 1084.0, + 393.0 + ], + "score": 0.99, + "text": "www.elsevier.com/locate/jhydrol" + }, + { + "category_id": 15, + "poly": [ + 134.0, + 1688.0, + 1377.0, + 1688.0, + 1377.0, + 1715.0, + 134.0, + 1715.0 + ], + "score": 0.99, + "text": "* Corresponding author. Address: Forest Science Centre, Department of Sustainability and Environment, P.O. Box 137, Heidelberg, Vic." + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1718.0, + 314.0, + 1718.0, + 314.0, + 1741.0, + 118.0, + 1741.0 + ], + "score": 0.99, + "text": "3084,Australia.Tel.:" + }, + { + "category_id": 15, + "poly": [ + 480.0, + 1718.0, + 701.0, + 1718.0, + 701.0, + 1741.0, + 480.0, + 1741.0 + ], + "score": 0.98, + "text": ";fax:+61394508644." + }, + { + "category_id": 15, + "poly": [ + 167.0, + 1748.0, + 655.0, + 1748.0, + 655.0, + 1768.0, + 167.0, + 1768.0 + ], + "score": 0.98, + "text": "-mailaddress:patrickl@unimelb.edu.au(P.N.J.Lane)." + }, + { + "category_id": 15, + "poly": [ + 211.0, + 623.0, + 1285.0, + 623.0, + 1285.0, + 653.0, + 211.0, + 653.0 + ], + "score": 0.97, + "text": "aSchool of Forest and EcosystemStudies,University ofMelbourne,P.O.Box 137,Heidelberg,Victoria 3084,Australia" + }, + { + "category_id": 15, + "poly": [ + 457.0, + 649.0, + 1038.0, + 649.0, + 1038.0, + 679.0, + 457.0, + 679.0 + ], + "score": 0.98, + "text": "bCSIRODivision of Land and Water,Canberra,ACT,Australia" + }, + { + "category_id": 15, + "poly": [ + 368.0, + 676.0, + 1127.0, + 676.0, + 1127.0, + 709.0, + 368.0, + 709.0 + ], + "score": 0.98, + "text": "cCooperative Research Centre for Catchment Hydrology, Canberra, ACT, Australia" + }, + { + "category_id": 15, + "poly": [ + 303.0, + 704.0, + 1198.0, + 704.0, + 1198.0, + 739.0, + 303.0, + 739.0 + ], + "score": 0.96, + "text": "Department of Civil and Environmental Engineering, University of Melbourne, Victoria, Australia" + }, + { + "category_id": 15, + "poly": [ + 247.0, + 475.0, + 1252.0, + 475.0, + 1252.0, + 518.0, + 247.0, + 518.0 + ], + "score": 0.99, + "text": "The response of flow duration curves to afforestation" + }, + { + "category_id": 15, + "poly": [ + 389.0, + 754.0, + 1107.0, + 754.0, + 1107.0, + 775.0, + 389.0, + 775.0 + ], + "score": 0.99, + "text": "Received1October 2003;revised22December2004;accepted3January2005" + }, + { + "category_id": 15, + "poly": [ + 141.0, + 914.0, + 1380.0, + 914.0, + 1380.0, + 944.0, + 141.0, + 944.0 + ], + "score": 0.99, + "text": "The hydrologic effect of replacing pasture or other short crops with trees is reasonably well understood on a mean annual" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 948.0, + 1377.0, + 948.0, + 1377.0, + 972.0, + 120.0, + 972.0 + ], + "score": 0.98, + "text": "basis. The impact on flow regime, as described by the annual flow duration curve (FDC) is less certain. A method to assess the" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 979.0, + 1379.0, + 979.0, + 1379.0, + 1003.0, + 120.0, + 1003.0 + ], + "score": 0.98, + "text": "impact of plantation establishment on FDCs was developed. The starting point for the analyses was the assumption that rainfall" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 1010.0, + 1379.0, + 1010.0, + 1379.0, + 1038.0, + 117.0, + 1038.0 + ], + "score": 1.0, + "text": "and vegetation age are the principal drivers of evapotranspiration. A key objective was to remove the variability in the rainfall" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1042.0, + 1377.0, + 1042.0, + 1377.0, + 1066.0, + 119.0, + 1066.0 + ], + "score": 0.98, + "text": "signal, leaving changes in streamflow solely attributable to the evapotranspiration of the plantation. A method was developed to" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1074.0, + 1379.0, + 1074.0, + 1379.0, + 1098.0, + 119.0, + 1098.0 + ], + "score": 0.97, + "text": "(1) fit a model to the observed annual time series of FDC percentiles; i.e. 1Oth percentile for each year of record with annual" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1106.0, + 1377.0, + 1106.0, + 1377.0, + 1130.0, + 119.0, + 1130.0 + ], + "score": 0.99, + "text": "rainfall and plantation age as parameters, (2) replace the annual rainfall variation with the long term mean to obtain climate" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1137.0, + 1379.0, + 1137.0, + 1379.0, + 1160.0, + 119.0, + 1160.0 + ], + "score": 0.96, + "text": "adjusted FDCs, and (3) quantify changes in FDC percentiles as plantations age. Data from 10 catchments from Australia, South" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 1165.0, + 1379.0, + 1165.0, + 1379.0, + 1194.0, + 117.0, + 1194.0 + ], + "score": 0.98, + "text": "Africa and New Zealand were used. The model was able to represent flow variation for the majority of percentiles at eight of the" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1199.0, + 1379.0, + 1199.0, + 1379.0, + 1223.0, + 120.0, + 1223.0 + ], + "score": 0.98, + "text": "10 catchments, particularly for the 10-5Oth percentiles. The adjusted FDCs revealed variable patterns in flow reductions with" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1232.0, + 1377.0, + 1232.0, + 1377.0, + 1255.0, + 119.0, + 1255.0 + ], + "score": 0.97, + "text": "two types of responses(groups)being identified.Group1 catchments show a substantial increase in the number of zeroflow" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1261.0, + 1377.0, + 1261.0, + 1377.0, + 1285.0, + 119.0, + 1285.0 + ], + "score": 0.98, + "text": "days, with low flows being more affected than high fows. Group 2 catchments show a more uniform reduction in flows across" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1294.0, + 1380.0, + 1294.0, + 1380.0, + 1318.0, + 119.0, + 1318.0 + ], + "score": 0.98, + "text": "all percentiles. The differences may be partly explained by storage characteristics. The modelled flow reductions were in accord" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1324.0, + 1382.0, + 1324.0, + 1382.0, + 1350.0, + 119.0, + 1350.0 + ], + "score": 0.99, + "text": "with published results of paired catchment experiments. An additional analysis was performed to characterise the impact of" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 1354.0, + 570.0, + 1354.0, + 570.0, + 1382.0, + 117.0, + 1382.0 + ], + "score": 0.95, + "text": "afforestation on thenumber ofzeroflowdays" + }, + { + "category_id": 15, + "poly": [ + 637.0, + 1354.0, + 1380.0, + 1354.0, + 1380.0, + 1382.0, + 637.0, + 1382.0 + ], + "score": 0.99, + "text": "for the catchments in group 1. This model performed particularly well, and" + }, + { + "category_id": 15, + "poly": [ + 116.0, + 1385.0, + 713.0, + 1385.0, + 713.0, + 1414.0, + 116.0, + 1414.0 + ], + "score": 0.99, + "text": "when adjusted for climate, indicated a significant increase in" + }, + { + "category_id": 15, + "poly": [ + 768.0, + 1385.0, + 1379.0, + 1385.0, + 1379.0, + 1414.0, + 768.0, + 1414.0 + ], + "score": 0.98, + "text": ".The zero flow day method could be used to determine change" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 1420.0, + 1379.0, + 1420.0, + 1379.0, + 1444.0, + 117.0, + 1444.0 + ], + "score": 0.99, + "text": "in the occurrence of any given flow in response to afforestation. The methods used in this study proved satisfactory in removing" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1452.0, + 1379.0, + 1452.0, + 1379.0, + 1476.0, + 119.0, + 1476.0 + ], + "score": 0.99, + "text": "the rainfall variability, and have added useful insight into the hydrologic impacts of plantation establishment. This approach" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1483.0, + 1376.0, + 1483.0, + 1376.0, + 1506.0, + 119.0, + 1506.0 + ], + "score": 0.97, + "text": "provides a methodologyfor understanding catchment response to afforestation,where paired catchment data is not available." + }, + { + "category_id": 15, + "poly": [ + 141.0, + 1512.0, + 536.0, + 1512.0, + 536.0, + 1531.0, + 141.0, + 1531.0 + ], + "score": 0.91, + "text": "2nn5FlcevierRVAllriohtsrecerved" + }, + { + "category_id": 15, + "poly": [ + 559.0, + 346.0, + 938.0, + 346.0, + 938.0, + 369.0, + 559.0, + 369.0 + ], + "score": 0.97, + "text": "Journalof Hydrology 310(2005)253-265" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 864.0, + 212.0, + 864.0, + 212.0, + 888.0, + 117.0, + 888.0 + ], + "score": 1.0, + "text": "Abstract" + }, + { + "category_id": 15, + "poly": [ + 235.0, + 547.0, + 1253.0, + 547.0, + 1253.0, + 608.0, + 235.0, + 608.0 + ], + "score": 0.94, + "text": "Patrick N.J. Laneac,*, Alice E. Bestb.c.d, Klaus Hickelb.c, Lu Zhangb." + }, + { + "category_id": 15, + "poly": [ + 117.0, + 1794.0, + 397.0, + 1794.0, + 397.0, + 1817.0, + 117.0, + 1817.0 + ], + "score": 0.97, + "text": "0022-1694/$ -see front matter" + }, + { + "category_id": 15, + "poly": [ + 420.0, + 1794.0, + 770.0, + 1794.0, + 770.0, + 1817.0, + 420.0, + 1817.0 + ], + "score": 0.98, + "text": "2005 Elsevier B.V.All rights reserved" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1824.0, + 422.0, + 1824.0, + 422.0, + 1842.0, + 119.0, + 1842.0 + ], + "score": 0.99, + "text": "doi:10.1016/j.jhydrol.2005.01.006" + }, + { + "category_id": 15, + "poly": [ + 517.0, + 193.0, + 984.0, + 193.0, + 984.0, + 220.0, + 517.0, + 220.0 + ], + "score": 0.98, + "text": "Available online atwww.sciencedirect.com" + }, + { + "category_id": 15, + "poly": [ + 601.0, + 248.0, + 723.0, + 248.0, + 723.0, + 269.0, + 601.0, + 269.0 + ], + "score": 0.99, + "text": "SCIENCE" + }, + { + "category_id": 15, + "poly": [ + 791.0, + 248.0, + 897.0, + 248.0, + 897.0, + 267.0, + 791.0, + 267.0 + ], + "score": 0.97, + "text": "OIRECT" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1572.0, + 865.0, + 1572.0, + 865.0, + 1595.0, + 118.0, + 1595.0 + ], + "score": 0.98, + "text": "Keywords: Afforestation; Flow duration curves; Flow reduction;Paired catchments" + }, + { + "category_id": 15, + "poly": [ + 141.0, + 1512.0, + 537.0, + 1512.0, + 537.0, + 1535.0, + 141.0, + 1535.0 + ], + "score": 0.98, + "text": "2005 Elsevier B.V.All rights reserved" + }, + { + "category_id": 15, + "poly": [ + 1084.0, + 374.0, + 1381.0, + 374.0, + 1381.0, + 396.0, + 1084.0, + 396.0 + ], + "score": 1.0, + "text": "www.elsevier.com/locate/jhydrol" + }, + { + "category_id": 15, + "poly": [ + 519.0, + 198.0, + 981.0, + 198.0, + 981.0, + 220.0, + 519.0, + 220.0 + ], + "score": 0.99, + "text": "Availableonlineatwww.sciencedirect.com" + } + ], + "page_info": { + "page_no": 0, + "height": 2064, + "width": 1512 + } + }, + { + "layout_dets": [ + { + "category_id": 4, + "poly": [ + 793.3238525390625, + 764.6008911132812, + 1394.982177734375, + 764.6008911132812, + 1394.982177734375, + 817.2584228515625, + 793.3238525390625, + 817.2584228515625 + ], + "score": 0.9999980926513672 + }, + { + "category_id": 1, + "poly": [ + 794.8445434570312, + 847.7525024414062, + 1396.7862548828125, + 847.7525024414062, + 1396.7862548828125, + 1280.268310546875, + 794.8445434570312, + 1280.268310546875 + ], + "score": 0.9999948740005493 + }, + { + "category_id": 1, + "poly": [ + 794.4091796875, + 1281.1240234375, + 1397.727783203125, + 1281.1240234375, + 1397.727783203125, + 1847.862060546875, + 794.4091796875, + 1847.862060546875 + ], + "score": 0.9999915957450867 + }, + { + "category_id": 3, + "poly": [ + 800.3482055664062, + 254.34396362304688, + 1385.85546875, + 254.34396362304688, + 1385.85546875, + 741.2379760742188, + 800.3482055664062, + 741.2379760742188 + ], + "score": 0.9999874830245972 + }, + { + "category_id": 1, + "poly": [ + 131.33212280273438, + 1017.2642211914062, + 731.5430297851562, + 1017.2642211914062, + 731.5430297851562, + 1848.0374755859375, + 131.33212280273438, + 1848.0374755859375 + ], + "score": 0.9999839663505554 + }, + { + "category_id": 1, + "poly": [ + 132.01792907714844, + 317.6101379394531, + 731.1533813476562, + 317.6101379394531, + 731.1533813476562, + 1015.9430541992188, + 132.01792907714844, + 1015.9430541992188 + ], + "score": 0.9999791979789734 + }, + { + "category_id": 0, + "poly": [ + 130.92127990722656, + 250.8260040283203, + 312.45001220703125, + 250.8260040283203, + 312.45001220703125, + 284.91973876953125, + 130.92127990722656, + 284.91973876953125 + ], + "score": 0.9999523162841797 + }, + { + "category_id": 2, + "poly": [ + 130.0436248779297, + 194.7170867919922, + 166.79232788085938, + 194.7170867919922, + 166.79232788085938, + 215.29795837402344, + 130.0436248779297, + 215.29795837402344 + ], + "score": 0.9999291300773621 + }, + { + "category_id": 2, + "poly": [ + 480.5660400390625, + 194.7841339111328, + 1045.443115234375, + 194.7841339111328, + 1045.443115234375, + 218.79908752441406, + 480.5660400390625, + 218.79908752441406 + ], + "score": 0.9998185038566589 + }, + { + "category_id": 13, + "poly": [ + 984, + 1180, + 1065, + 1180, + 1065, + 1211, + 984, + 1211 + ], + "score": 0.88, + "latex": "<\\!20\\%" + }, + { + "category_id": 13, + "poly": [ + 128, + 1415, + 183, + 1415, + 183, + 1445, + 128, + 1445 + ], + "score": 0.86, + "latex": "95\\%" + }, + { + "category_id": 13, + "poly": [ + 573, + 618, + 723, + 618, + 723, + 649, + 573, + 649 + ], + "score": 0.67, + "latex": "400{-}500\\ \\mathrm{mm}" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 768.0, + 1390.0, + 768.0, + 1390.0, + 790.0, + 796.0, + 790.0 + ], + "score": 0.96, + "text": "Fig.1.Annual flow duration curves of daily flows from Pine Creek" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 796.0, + 993.0, + 796.0, + 993.0, + 815.0, + 796.0, + 815.0 + ], + "score": 0.99, + "text": "Australia,1989-2000." + }, + { + "category_id": 15, + "poly": [ + 796.0, + 853.0, + 1392.0, + 853.0, + 1392.0, + 877.0, + 796.0, + 877.0 + ], + "score": 0.97, + "text": "apply on a seasonal or shorter scale.Further, the" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 886.0, + 1391.0, + 886.0, + 1391.0, + 912.0, + 797.0, + 912.0 + ], + "score": 0.98, + "text": "observed impacts of any land use change on flows may" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 920.0, + 1391.0, + 920.0, + 1391.0, + 944.0, + 796.0, + 944.0 + ], + "score": 0.96, + "text": "beexaggerated or understated depending onthe" + }, + { + "category_id": 15, + "poly": [ + 794.0, + 952.0, + 1393.0, + 952.0, + 1393.0, + 978.0, + 794.0, + 978.0 + ], + "score": 0.96, + "text": "prevailing climate.Observationsof flowduring" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 987.0, + 1392.0, + 987.0, + 1392.0, + 1011.0, + 796.0, + 1011.0 + ], + "score": 0.95, + "text": "extended wet or dry spells, or with high annual" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 1021.0, + 1392.0, + 1021.0, + 1392.0, + 1045.0, + 797.0, + 1045.0 + ], + "score": 0.99, + "text": "variability can obscure the real impacts. Fig. 1 plots" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1053.0, + 1392.0, + 1053.0, + 1392.0, + 1077.0, + 796.0, + 1077.0 + ], + "score": 0.97, + "text": "annual FDCs over12years ofplantationgrowthforone" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 1086.0, + 1391.0, + 1086.0, + 1391.0, + 1109.0, + 797.0, + 1109.0 + ], + "score": 0.97, + "text": "of thecatchmentsusedin thisstudy,Pine Creek.The" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1120.0, + 1390.0, + 1120.0, + 1390.0, + 1145.0, + 796.0, + 1145.0 + ], + "score": 0.94, + "text": "net change inflow is obscured byrainfallvariability:" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1155.0, + 1391.0, + 1155.0, + 1391.0, + 1175.0, + 796.0, + 1175.0 + ], + "score": 0.99, + "text": "e.g.thegreatestchangeintheFDCisin1996,withthe" + }, + { + "category_id": 15, + "poly": [ + 798.0, + 1187.0, + 983.0, + 1187.0, + 983.0, + 1210.0, + 798.0, + 1210.0 + ], + "score": 0.99, + "text": "streamflowing" + }, + { + "category_id": 15, + "poly": [ + 1066.0, + 1187.0, + 1391.0, + 1187.0, + 1391.0, + 1210.0, + 1066.0, + 1210.0 + ], + "score": 1.0, + "text": "ofthetime.Thismaybe" + }, + { + "category_id": 15, + "poly": [ + 798.0, + 1221.0, + 1389.0, + 1221.0, + 1389.0, + 1244.0, + 798.0, + 1244.0 + ], + "score": 0.96, + "text": "comparedwith2o00,wherethereissubstantially" + }, + { + "category_id": 15, + "poly": [ + 795.0, + 1250.0, + 940.0, + 1250.0, + 940.0, + 1281.0, + 795.0, + 1281.0 + ], + "score": 1.0, + "text": "higherflows." + }, + { + "category_id": 15, + "poly": [ + 832.0, + 1287.0, + 1393.0, + 1287.0, + 1393.0, + 1310.0, + 832.0, + 1310.0 + ], + "score": 0.98, + "text": "Thispaperpresentstheresultsof aproject aimed at" + }, + { + "category_id": 15, + "poly": [ + 798.0, + 1321.0, + 1394.0, + 1321.0, + 1394.0, + 1345.0, + 798.0, + 1345.0 + ], + "score": 0.99, + "text": "quantifyingchangesinannualflowregimeof" + }, + { + "category_id": 15, + "poly": [ + 795.0, + 1353.0, + 1393.0, + 1353.0, + 1393.0, + 1380.0, + 795.0, + 1380.0 + ], + "score": 0.98, + "text": "catchments following plantation establishment. The" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 1389.0, + 1392.0, + 1389.0, + 1392.0, + 1411.0, + 797.0, + 1411.0 + ], + "score": 0.99, + "text": "flowregimeisrepresentedbytheflowdurationcurve" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 1420.0, + 1393.0, + 1420.0, + 1393.0, + 1443.0, + 797.0, + 1443.0 + ], + "score": 1.0, + "text": "(FDC).Thekeyassumptionwasthatrainfalland" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1453.0, + 1391.0, + 1453.0, + 1391.0, + 1479.0, + 796.0, + 1479.0 + ], + "score": 0.98, + "text": "forest age are the principal drivers of evapotranspira-" + }, + { + "category_id": 15, + "poly": [ + 795.0, + 1487.0, + 1394.0, + 1487.0, + 1394.0, + 1513.0, + 795.0, + 1513.0 + ], + "score": 0.97, + "text": "tion. For any generalisation of response of the FDC to" + }, + { + "category_id": 15, + "poly": [ + 799.0, + 1521.0, + 1393.0, + 1521.0, + 1393.0, + 1545.0, + 799.0, + 1545.0 + ], + "score": 0.98, + "text": "vegetation change, the variation in the annual climate" + }, + { + "category_id": 15, + "poly": [ + 798.0, + 1554.0, + 1393.0, + 1554.0, + 1393.0, + 1577.0, + 798.0, + 1577.0 + ], + "score": 1.0, + "text": "signalmustberemoved.Thetime-testedsolutionto" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 1588.0, + 1393.0, + 1588.0, + 1393.0, + 1612.0, + 797.0, + 1612.0 + ], + "score": 0.99, + "text": "this problem is the paired-catchment (control versus" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1620.0, + 1393.0, + 1620.0, + 1393.0, + 1645.0, + 796.0, + 1645.0 + ], + "score": 0.96, + "text": "treatment)experiment. The benefits in such studies" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1654.0, + 1391.0, + 1654.0, + 1391.0, + 1677.0, + 796.0, + 1677.0 + ], + "score": 1.0, + "text": "aremanifold:unambiguousmeasuresoftrends," + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1688.0, + 1391.0, + 1688.0, + 1391.0, + 1712.0, + 796.0, + 1712.0 + ], + "score": 0.96, + "text": "insights intothe processes driving those trends," + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1722.0, + 1391.0, + 1722.0, + 1391.0, + 1744.0, + 796.0, + 1744.0 + ], + "score": 1.0, + "text": "excellentopportunitiesformodelparameterisation" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1752.0, + 1390.0, + 1752.0, + 1390.0, + 1777.0, + 796.0, + 1777.0 + ], + "score": 1.0, + "text": "andvalidation.Howeverthesedataarenotreadily" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 1787.0, + 1391.0, + 1787.0, + 1391.0, + 1811.0, + 797.0, + 1811.0 + ], + "score": 0.95, + "text": "available for the range of treamtments and environ-" + }, + { + "category_id": 15, + "poly": [ + 798.0, + 1822.0, + 1391.0, + 1822.0, + 1391.0, + 1845.0, + 798.0, + 1845.0 + ], + "score": 0.97, + "text": "ments required.Consequently, the aims of this project" + }, + { + "category_id": 15, + "poly": [ + 166.0, + 1022.0, + 728.0, + 1022.0, + 728.0, + 1047.0, + 166.0, + 1047.0 + ], + "score": 0.96, + "text": "Zhang et al.(1999,2001)developed simple and" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 1057.0, + 728.0, + 1057.0, + 728.0, + 1081.0, + 133.0, + 1081.0 + ], + "score": 0.95, + "text": "easily parameterised models topredict changes in" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1089.0, + 728.0, + 1089.0, + 728.0, + 1113.0, + 132.0, + 1113.0 + ], + "score": 0.98, + "text": "mean annual flows following afforestation. However," + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1121.0, + 727.0, + 1121.0, + 727.0, + 1147.0, + 132.0, + 1147.0 + ], + "score": 0.98, + "text": "there is a need to consider the annual flow regime as the" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 1158.0, + 727.0, + 1158.0, + 727.0, + 1181.0, + 133.0, + 1181.0 + ], + "score": 1.0, + "text": "relativechangesinhighandlowflowsmayhave" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1189.0, + 727.0, + 1189.0, + 727.0, + 1214.0, + 131.0, + 1214.0 + ], + "score": 0.97, + "text": "considerable site specific and downstream impacts." + }, + { + "category_id": 15, + "poly": [ + 133.0, + 1221.0, + 728.0, + 1221.0, + 728.0, + 1246.0, + 133.0, + 1246.0 + ], + "score": 0.94, + "text": "Sikka et al.(2003)recently showed a change from" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1256.0, + 728.0, + 1256.0, + 728.0, + 1279.0, + 132.0, + 1279.0 + ], + "score": 1.0, + "text": "grasslandtoEucalyptusglobulusplantationsinIndia" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1288.0, + 726.0, + 1288.0, + 726.0, + 1313.0, + 132.0, + 1313.0 + ], + "score": 0.99, + "text": "decreasedalowflowindexbyafactoroftwoduringthe" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1321.0, + 729.0, + 1321.0, + 729.0, + 1346.0, + 131.0, + 1346.0 + ], + "score": 0.96, + "text": "first rotation (9 years), and by 3.75 during the second" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 1355.0, + 727.0, + 1355.0, + 727.0, + 1379.0, + 133.0, + 1379.0 + ], + "score": 0.98, + "text": "rotation, with more subdued impact on peak flows. The" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1387.0, + 729.0, + 1387.0, + 729.0, + 1414.0, + 131.0, + 1414.0 + ], + "score": 0.97, + "text": "index was defined as the 10 day average flow exceeded" + }, + { + "category_id": 15, + "poly": [ + 184.0, + 1421.0, + 727.0, + 1421.0, + 727.0, + 1446.0, + 184.0, + 1446.0 + ], + "score": 0.96, + "text": "of thetime,obtained from analysis of10-dayflow" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1454.0, + 729.0, + 1454.0, + 729.0, + 1480.0, + 132.0, + 1480.0 + ], + "score": 0.95, + "text": "duration curves.Scott and Smith (1997) reported" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1488.0, + 729.0, + 1488.0, + 729.0, + 1515.0, + 131.0, + 1515.0 + ], + "score": 0.98, + "text": "proportionallygreater reductionsinlowflows" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 1522.0, + 729.0, + 1522.0, + 729.0, + 1545.0, + 133.0, + 1545.0 + ], + "score": 0.97, + "text": "(75-1oothpercentiles)thanannualflowsfromSouth" + }, + { + "category_id": 15, + "poly": [ + 135.0, + 1556.0, + 728.0, + 1556.0, + 728.0, + 1576.0, + 135.0, + 1576.0 + ], + "score": 1.0, + "text": "Africanresearchcatchmentsunderconversionsfrom" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1590.0, + 729.0, + 1590.0, + 729.0, + 1610.0, + 132.0, + 1610.0 + ], + "score": 0.99, + "text": "grasstopineandeucalyptplantations,whileBosch" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 1621.0, + 727.0, + 1621.0, + 727.0, + 1644.0, + 133.0, + 1644.0 + ], + "score": 0.99, + "text": "(1979)foundthegreatestreductioninseasonalflow" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1653.0, + 727.0, + 1653.0, + 727.0, + 1678.0, + 131.0, + 1678.0 + ], + "score": 1.0, + "text": "fromthesummerwetseason.FaheyandJackson" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 1687.0, + 727.0, + 1687.0, + 727.0, + 1710.0, + 133.0, + 1710.0 + ], + "score": 0.99, + "text": "(1997)reportedthereductioninpeakflowswastwice" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1719.0, + 727.0, + 1719.0, + 727.0, + 1742.0, + 132.0, + 1742.0 + ], + "score": 0.99, + "text": "thatoftotalflowandlowflowsforpine afforestationin" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1755.0, + 728.0, + 1755.0, + 728.0, + 1778.0, + 132.0, + 1778.0 + ], + "score": 1.0, + "text": "NewZealand.Thegeneralisationsthatcanbedrawn" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1785.0, + 727.0, + 1785.0, + 727.0, + 1813.0, + 131.0, + 1813.0 + ], + "score": 1.0, + "text": "from annual analyses, where processes and hydrologic" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 1822.0, + 728.0, + 1822.0, + 728.0, + 1845.0, + 133.0, + 1845.0 + ], + "score": 1.0, + "text": "responsesaretoacertainextentintegratedmaynot" + }, + { + "category_id": 15, + "poly": [ + 166.0, + 322.0, + 728.0, + 322.0, + 728.0, + 350.0, + 166.0, + 350.0 + ], + "score": 0.98, + "text": "Widespread afforestation through plantation estab-" + }, + { + "category_id": 15, + "poly": [ + 129.0, + 355.0, + 726.0, + 355.0, + 726.0, + 384.0, + 129.0, + 384.0 + ], + "score": 0.99, + "text": "lishment on non-forested land represents a potentially" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 391.0, + 726.0, + 391.0, + 726.0, + 416.0, + 133.0, + 416.0 + ], + "score": 1.0, + "text": "significantalterationofcatchmentevapotranspiration" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 426.0, + 728.0, + 426.0, + 728.0, + 449.0, + 133.0, + 449.0 + ], + "score": 1.0, + "text": "(ET).Usingdatacollatedfrommultiplecatchment" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 458.0, + 729.0, + 458.0, + 729.0, + 480.0, + 133.0, + 480.0 + ], + "score": 1.0, + "text": "studies,researchershavedemonstratedaconsistent" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 489.0, + 729.0, + 489.0, + 729.0, + 514.0, + 132.0, + 514.0 + ], + "score": 0.95, + "text": "difference in ET between forests and grass or short" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 524.0, + 729.0, + 524.0, + 729.0, + 546.0, + 133.0, + 546.0 + ], + "score": 0.99, + "text": "crops,andtherelationshipbetweenETandrainfallon" + }, + { + "category_id": 15, + "poly": [ + 129.0, + 557.0, + 728.0, + 557.0, + 728.0, + 579.0, + 129.0, + 579.0 + ], + "score": 0.96, + "text": "a meanannualbasis (Holmes andSinclair,1986;" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 588.0, + 727.0, + 588.0, + 727.0, + 616.0, + 131.0, + 616.0 + ], + "score": 0.95, + "text": "Vertessy and Bessard, 1999; Zhang et al., 1999," + }, + { + "category_id": 15, + "poly": [ + 132.0, + 622.0, + 572.0, + 622.0, + 572.0, + 648.0, + 132.0, + 648.0 + ], + "score": 0.96, + "text": "2001).Once annual rainfall exceeds" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 657.0, + 728.0, + 657.0, + 728.0, + 681.0, + 132.0, + 681.0 + ], + "score": 0.99, + "text": "there is an increasing divergence between forest and" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 691.0, + 728.0, + 691.0, + 728.0, + 713.0, + 131.0, + 713.0 + ], + "score": 0.97, + "text": "grasslandET(Zhangetal.,20o1).Researchfrom" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 724.0, + 728.0, + 724.0, + 728.0, + 748.0, + 133.0, + 748.0 + ], + "score": 0.96, + "text": "SouthAfrica inparticular has demonstratedflow" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 758.0, + 729.0, + 758.0, + 729.0, + 781.0, + 133.0, + 781.0 + ], + "score": 1.0, + "text": "reductionfollowingafforestationwithbothpineand" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 791.0, + 728.0, + 791.0, + 728.0, + 813.0, + 132.0, + 813.0 + ], + "score": 0.98, + "text": "eucalyptspecies(Bosch,1979;VanLilletal.,1980;" + }, + { + "category_id": 15, + "poly": [ + 134.0, + 824.0, + 729.0, + 824.0, + 729.0, + 847.0, + 134.0, + 847.0 + ], + "score": 0.99, + "text": "VanWyk,1987;BoschandVonGadow,1990;Scott" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 856.0, + 727.0, + 856.0, + 727.0, + 880.0, + 132.0, + 880.0 + ], + "score": 0.99, + "text": "and Smith, 1997; Scott et al., 2000). In regions, where" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 891.0, + 728.0, + 891.0, + 728.0, + 914.0, + 133.0, + 914.0 + ], + "score": 1.0, + "text": "waterisanincreasinglyvaluableresource,prediction" + }, + { + "category_id": 15, + "poly": [ + 134.0, + 922.0, + 727.0, + 922.0, + 727.0, + 946.0, + 134.0, + 946.0 + ], + "score": 0.97, + "text": "of the long-term hydrologic impact of afforestation is" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 957.0, + 728.0, + 957.0, + 728.0, + 980.0, + 132.0, + 980.0 + ], + "score": 1.0, + "text": "aprerequisitefortheoptimalplanningofcatchment" + }, + { + "category_id": 15, + "poly": [ + 130.0, + 987.0, + 232.0, + 987.0, + 232.0, + 1013.0, + 130.0, + 1013.0 + ], + "score": 1.0, + "text": "landuse." + }, + { + "category_id": 15, + "poly": [ + 130.0, + 253.0, + 310.0, + 253.0, + 310.0, + 282.0, + 130.0, + 282.0 + ], + "score": 1.0, + "text": "1. Introduction" + }, + { + "category_id": 15, + "poly": [ + 129.0, + 195.0, + 167.0, + 195.0, + 167.0, + 219.0, + 129.0, + 219.0 + ], + "score": 1.0, + "text": "254" + }, + { + "category_id": 15, + "poly": [ + 482.0, + 197.0, + 1040.0, + 197.0, + 1040.0, + 220.0, + 482.0, + 220.0 + ], + "score": 0.99, + "text": "P.N.J.Laneet al./JournalofHydrology310(2005)253-265" + } + ], + "page_info": { + "page_no": 1, + "height": 2064, + "width": 1512 + } + }, + { + "layout_dets": [ + { + "category_id": 0, + "poly": [ + 117.98583221435547, + 651.4143676757812, + 250.4243927001953, + 651.4143676757812, + 250.4243927001953, + 681.6177978515625, + 117.98583221435547, + 681.6177978515625 + ], + "score": 0.9999960660934448 + }, + { + "category_id": 1, + "poly": [ + 117.66859436035156, + 252.15028381347656, + 717.0852661132812, + 252.15028381347656, + 717.0852661132812, + 583.2984619140625, + 117.66859436035156, + 583.2984619140625 + ], + "score": 0.9999945163726807 + }, + { + "category_id": 1, + "poly": [ + 781.6583251953125, + 254.4632568359375, + 1380.35498046875, + 254.4632568359375, + 1380.35498046875, + 383.93988037109375, + 781.6583251953125, + 383.93988037109375 + ], + "score": 0.9999942779541016 + }, + { + "category_id": 1, + "poly": [ + 117.45108795166016, + 787.5255126953125, + 717.2050170898438, + 787.5255126953125, + 717.2050170898438, + 1283.052001953125, + 117.45108795166016, + 1283.052001953125 + ], + "score": 0.9999866485595703 + }, + { + "category_id": 1, + "poly": [ + 781.4449462890625, + 518.4725952148438, + 1381.476806640625, + 518.4725952148438, + 1381.476806640625, + 1115.634033203125, + 781.4449462890625, + 1115.634033203125 + ], + "score": 0.9999841451644897 + }, + { + "category_id": 1, + "poly": [ + 117.63248443603516, + 1283.588134765625, + 717.8546752929688, + 1283.588134765625, + 717.8546752929688, + 1846.96337890625, + 117.63248443603516, + 1846.96337890625 + ], + "score": 0.9999801516532898 + }, + { + "category_id": 1, + "poly": [ + 781.0757446289062, + 1281.677490234375, + 1381.9857177734375, + 1281.677490234375, + 1381.9857177734375, + 1846.426025390625, + 781.0757446289062, + 1846.426025390625 + ], + "score": 0.9999785423278809 + }, + { + "category_id": 0, + "poly": [ + 118.35875701904297, + 719.4396362304688, + 523.7080688476562, + 719.4396362304688, + 523.7080688476562, + 748.3139038085938, + 118.35875701904297, + 748.3139038085938 + ], + "score": 0.999972939491272 + }, + { + "category_id": 2, + "poly": [ + 1346.603759765625, + 195.0556182861328, + 1381.0526123046875, + 195.0556182861328, + 1381.0526123046875, + 216.44448852539062, + 1346.603759765625, + 216.44448852539062 + ], + "score": 0.9999720454216003 + }, + { + "category_id": 9, + "poly": [ + 1346.9273681640625, + 438.05438232421875, + 1379.6627197265625, + 438.05438232421875, + 1379.6627197265625, + 465.9045104980469, + 1346.9273681640625, + 465.9045104980469 + ], + "score": 0.999956488609314 + }, + { + "category_id": 8, + "poly": [ + 776.7713012695312, + 1153.820556640625, + 1201.1727294921875, + 1153.820556640625, + 1201.1727294921875, + 1238.2696533203125, + 776.7713012695312, + 1238.2696533203125 + ], + "score": 0.9999523162841797 + }, + { + "category_id": 9, + "poly": [ + 1347.7716064453125, + 1178.136474609375, + 1379.205810546875, + 1178.136474609375, + 1379.205810546875, + 1209.233154296875, + 1347.7716064453125, + 1209.233154296875 + ], + "score": 0.9999113082885742 + }, + { + "category_id": 2, + "poly": [ + 466.5557861328125, + 194.43609619140625, + 1031.927490234375, + 194.43609619140625, + 1031.927490234375, + 219.32997131347656, + 466.5557861328125, + 219.32997131347656 + ], + "score": 0.9998723864555359 + }, + { + "category_id": 8, + "poly": [ + 779.8624877929688, + 430.4825439453125, + 996.8544311523438, + 430.4825439453125, + 996.8544311523438, + 471.1011047363281, + 779.8624877929688, + 471.1011047363281 + ], + "score": 0.9997532367706299 + }, + { + "category_id": 14, + "poly": [ + 777, + 1156, + 1200, + 1156, + 1200, + 1237, + 777, + 1237 + ], + "score": 0.92, + "latex": "Q_{\\mathcal{U}}=a+b(\\Delta P)+\\frac{Y}{1+\\exp\\left(\\frac{T-T_{\\mathrm{half}}}{S}\\right)}" + }, + { + "category_id": 13, + "poly": [ + 1150, + 520, + 1201, + 520, + 1201, + 551, + 1150, + 551 + ], + "score": 0.9, + "latex": "f(P)" + }, + { + "category_id": 13, + "poly": [ + 1210, + 1384, + 1262, + 1384, + 1262, + 1414, + 1210, + 1414 + ], + "score": 0.9, + "latex": "T_{\\mathrm{half}}" + }, + { + "category_id": 13, + "poly": [ + 856, + 520, + 897, + 520, + 897, + 550, + 856, + 550 + ], + "score": 0.9, + "latex": "Q_{\\%}" + }, + { + "category_id": 13, + "poly": [ + 930, + 552, + 982, + 552, + 982, + 584, + 930, + 584 + ], + "score": 0.89, + "latex": "g(T)" + }, + { + "category_id": 13, + "poly": [ + 857, + 1285, + 898, + 1285, + 898, + 1315, + 857, + 1315 + ], + "score": 0.89, + "latex": "Q_{\\%}" + }, + { + "category_id": 13, + "poly": [ + 1196, + 1649, + 1278, + 1649, + 1278, + 1678, + 1196, + 1678 + ], + "score": 0.89, + "latex": "\\Delta P\\!=\\!0" + }, + { + "category_id": 13, + "poly": [ + 1270, + 1483, + 1311, + 1483, + 1311, + 1515, + 1270, + 1515 + ], + "score": 0.89, + "latex": "Q_{\\%}" + }, + { + "category_id": 13, + "poly": [ + 1259, + 1418, + 1301, + 1418, + 1301, + 1449, + 1259, + 1449 + ], + "score": 0.89, + "latex": "Q_{\\%}" + }, + { + "category_id": 13, + "poly": [ + 1075, + 1682, + 1140, + 1682, + 1140, + 1711, + 1075, + 1711 + ], + "score": 0.88, + "latex": "a\\!+\\!Y." + }, + { + "category_id": 13, + "poly": [ + 895, + 1483, + 976, + 1483, + 976, + 1512, + 895, + 1512 + ], + "score": 0.88, + "latex": "\\Delta P\\!=\\!0" + }, + { + "category_id": 13, + "poly": [ + 1206, + 1285, + 1252, + 1285, + 1252, + 1315, + 1206, + 1315 + ], + "score": 0.88, + "latex": "Q_{50}" + }, + { + "category_id": 13, + "poly": [ + 779, + 1682, + 821, + 1682, + 821, + 1714, + 779, + 1714 + ], + "score": 0.88, + "latex": "Q_{\\%}" + }, + { + "category_id": 13, + "poly": [ + 1313, + 1649, + 1374, + 1649, + 1374, + 1678, + 1313, + 1678 + ], + "score": 0.87, + "latex": "T\\!=\\!0" + }, + { + "category_id": 14, + "poly": [ + 777, + 432, + 997, + 432, + 997, + 470, + 777, + 470 + ], + "score": 0.83, + "latex": "Q_{\\mathcal{I}_{\\theta}}=f(P)+g(T)" + }, + { + "category_id": 13, + "poly": [ + 963, + 1350, + 1002, + 1350, + 1002, + 1378, + 963, + 1378 + ], + "score": 0.8, + "latex": "\\Delta P" + }, + { + "category_id": 13, + "poly": [ + 989, + 1318, + 1012, + 1318, + 1012, + 1345, + 989, + 1345 + ], + "score": 0.64, + "latex": "Y" + }, + { + "category_id": 13, + "poly": [ + 1077, + 1318, + 1098, + 1318, + 1098, + 1345, + 1077, + 1345 + ], + "score": 0.64, + "latex": "S" + }, + { + "category_id": 13, + "poly": [ + 1239, + 1583, + 1262, + 1583, + 1262, + 1611, + 1239, + 1611 + ], + "score": 0.51, + "latex": "S" + }, + { + "category_id": 13, + "poly": [ + 989, + 1488, + 1008, + 1488, + 1008, + 1511, + 989, + 1511 + ], + "score": 0.3, + "latex": "a" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 655.0, + 250.0, + 655.0, + 250.0, + 679.0, + 117.0, + 679.0 + ], + "score": 1.0, + "text": "2. Methods" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 259.0, + 713.0, + 259.0, + 713.0, + 282.0, + 119.0, + 282.0 + ], + "score": 0.97, + "text": "wereto(1)fitamodeltotheobservedannualtime" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 293.0, + 715.0, + 293.0, + 715.0, + 317.0, + 119.0, + 317.0 + ], + "score": 0.97, + "text": "series of FDC percentiles; i.e. 10th percentile for each" + }, + { + "category_id": 15, + "poly": [ + 115.0, + 323.0, + 714.0, + 323.0, + 714.0, + 353.0, + 115.0, + 353.0 + ], + "score": 0.99, + "text": "year of record with annual rainfall and plantation age" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 360.0, + 715.0, + 360.0, + 715.0, + 382.0, + 118.0, + 382.0 + ], + "score": 0.99, + "text": "asparameters,(2)replacetheannualrainfallvariation" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 393.0, + 715.0, + 393.0, + 715.0, + 416.0, + 120.0, + 416.0 + ], + "score": 1.0, + "text": "withthelongtermmeantoobtainclimateadjusted" + }, + { + "category_id": 15, + "poly": [ + 116.0, + 423.0, + 716.0, + 423.0, + 716.0, + 452.0, + 116.0, + 452.0 + ], + "score": 0.99, + "text": "FDCs, and (3) quantify changes in FDC percentiles as" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 460.0, + 714.0, + 460.0, + 714.0, + 485.0, + 118.0, + 485.0 + ], + "score": 0.99, + "text": "plantations age. If the climate signal, represented by" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 490.0, + 714.0, + 490.0, + 714.0, + 517.0, + 117.0, + 517.0 + ], + "score": 0.98, + "text": "rainfall, could be successfully removed, the resulting" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 522.0, + 715.0, + 522.0, + 715.0, + 551.0, + 118.0, + 551.0 + ], + "score": 0.98, + "text": "changes in the FDC would be solely attributable to the" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 558.0, + 238.0, + 558.0, + 238.0, + 583.0, + 119.0, + 583.0 + ], + "score": 0.99, + "text": "vegetation." + }, + { + "category_id": 15, + "poly": [ + 785.0, + 259.0, + 1377.0, + 259.0, + 1377.0, + 283.0, + 785.0, + 283.0 + ], + "score": 0.97, + "text": "closure, a time term is required to represent plantation" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 293.0, + 1380.0, + 293.0, + 1380.0, + 317.0, + 782.0, + 317.0 + ], + "score": 0.96, + "text": "growth.A simple model relating the time series of" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 323.0, + 1379.0, + 323.0, + 1379.0, + 351.0, + 782.0, + 351.0 + ], + "score": 0.98, + "text": "each decile with rainfall and vegetation characteristics" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 355.0, + 1011.0, + 355.0, + 1011.0, + 385.0, + 782.0, + 385.0 + ], + "score": 0.98, + "text": "can be expressed as:" + }, + { + "category_id": 15, + "poly": [ + 150.0, + 789.0, + 713.0, + 789.0, + 713.0, + 816.0, + 150.0, + 816.0 + ], + "score": 0.99, + "text": "Flowdurationcurvesdisplaythe relationship" + }, + { + "category_id": 15, + "poly": [ + 116.0, + 822.0, + 715.0, + 822.0, + 715.0, + 850.0, + 116.0, + 850.0 + ], + "score": 0.97, + "text": "between streamflowand the percentage of time" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 856.0, + 712.0, + 856.0, + 712.0, + 881.0, + 117.0, + 881.0 + ], + "score": 1.0, + "text": "thestreamflowisexceededasacumulativedensity" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 888.0, + 715.0, + 888.0, + 715.0, + 915.0, + 117.0, + 915.0 + ], + "score": 0.99, + "text": "function They can be constructed for any time period" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 922.0, + 715.0, + 922.0, + 715.0, + 948.0, + 118.0, + 948.0 + ], + "score": 0.99, + "text": "(daily, weekly, monthly, etc.) and provide a graphical" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 956.0, + 711.0, + 956.0, + 711.0, + 981.0, + 118.0, + 981.0 + ], + "score": 0.99, + "text": "andstatisticalviewofhistoricstreamflowvariability" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 991.0, + 714.0, + 991.0, + 714.0, + 1014.0, + 117.0, + 1014.0 + ], + "score": 1.0, + "text": "inasinglecatchmentoracomparisonofinter-" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 1022.0, + 715.0, + 1022.0, + 715.0, + 1048.0, + 117.0, + 1048.0 + ], + "score": 0.98, + "text": "catchment flow regimes. Vogel and Fennessey (1994)" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1055.0, + 712.0, + 1055.0, + 712.0, + 1080.0, + 118.0, + 1080.0 + ], + "score": 0.96, + "text": "and Smakhtin(1999,2001)demonstrate the utility" + }, + { + "category_id": 15, + "poly": [ + 116.0, + 1088.0, + 714.0, + 1088.0, + 714.0, + 1116.0, + 116.0, + 1116.0 + ], + "score": 0.98, + "text": "(and caveats) of FDCs in characterising, comparing" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1123.0, + 715.0, + 1123.0, + 715.0, + 1150.0, + 119.0, + 1150.0 + ], + "score": 0.96, + "text": "and predicting flow regimes at varying temporal" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 1155.0, + 716.0, + 1155.0, + 716.0, + 1180.0, + 117.0, + 1180.0 + ], + "score": 0.98, + "text": "scales.Fig.1isanexampleof annualFDCs" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1190.0, + 712.0, + 1190.0, + 712.0, + 1212.0, + 119.0, + 1212.0 + ], + "score": 1.0, + "text": "constructedfromdailyflows.Fortheconsideration" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 1220.0, + 714.0, + 1220.0, + 714.0, + 1248.0, + 117.0, + 1248.0 + ], + "score": 0.99, + "text": "of annual flow regime, daily fows are an appropriate" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1257.0, + 474.0, + 1257.0, + 474.0, + 1280.0, + 119.0, + 1280.0 + ], + "score": 1.0, + "text": "timestepforFDCconstruction." + }, + { + "category_id": 15, + "poly": [ + 787.0, + 524.0, + 855.0, + 524.0, + 855.0, + 548.0, + 787.0, + 548.0 + ], + "score": 1.0, + "text": "where" + }, + { + "category_id": 15, + "poly": [ + 898.0, + 524.0, + 1149.0, + 524.0, + 1149.0, + 548.0, + 898.0, + 548.0 + ], + "score": 0.99, + "text": "is the percentile flow," + }, + { + "category_id": 15, + "poly": [ + 1202.0, + 524.0, + 1380.0, + 524.0, + 1380.0, + 548.0, + 1202.0, + 548.0 + ], + "score": 0.97, + "text": "is a function of" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 558.0, + 929.0, + 558.0, + 929.0, + 581.0, + 783.0, + 581.0 + ], + "score": 1.0, + "text": "rainfalland" + }, + { + "category_id": 15, + "poly": [ + 983.0, + 558.0, + 1378.0, + 558.0, + 1378.0, + 581.0, + 983.0, + 581.0 + ], + "score": 1.0, + "text": "isafunctionoftheageofthe" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 590.0, + 1379.0, + 590.0, + 1379.0, + 613.0, + 783.0, + 613.0 + ], + "score": 1.0, + "text": "plantation.Annualrainfallwaschosenastherainfall" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 624.0, + 1380.0, + 624.0, + 1380.0, + 647.0, + 784.0, + 647.0 + ], + "score": 1.0, + "text": "statisticasitprovedtobethemostrobustpredictorof" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 656.0, + 1379.0, + 656.0, + 1379.0, + 682.0, + 783.0, + 682.0 + ], + "score": 0.95, + "text": "flow over the whole range of flow percentiles,as" + }, + { + "category_id": 15, + "poly": [ + 785.0, + 692.0, + 1379.0, + 692.0, + 1379.0, + 716.0, + 785.0, + 716.0 + ], + "score": 1.0, + "text": "compared with rainfall percentiles; e.g. median rain-" + }, + { + "category_id": 15, + "poly": [ + 781.0, + 720.0, + 1380.0, + 720.0, + 1380.0, + 748.0, + 781.0, + 748.0 + ], + "score": 0.96, + "text": "fall versus 1oth flow percentile. The use of annual" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 757.0, + 1378.0, + 757.0, + 1378.0, + 781.0, + 783.0, + 781.0 + ], + "score": 0.97, + "text": "rainfall also minimises parameter complexity. The" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 788.0, + 1379.0, + 788.0, + 1379.0, + 815.0, + 783.0, + 815.0 + ], + "score": 0.98, + "text": "choice of model form is dependent on selecting a" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 822.0, + 1379.0, + 822.0, + 1379.0, + 846.0, + 782.0, + 846.0 + ], + "score": 0.98, + "text": "function that describes the relationship betweenforest" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 856.0, + 1381.0, + 856.0, + 1381.0, + 880.0, + 782.0, + 880.0 + ], + "score": 0.96, + "text": "age and ET.Scott and Smith (1997) demonstrated" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 890.0, + 1379.0, + 890.0, + 1379.0, + 913.0, + 784.0, + 913.0 + ], + "score": 0.99, + "text": "cumulativereductionsinannualandlowflows" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 922.0, + 1379.0, + 922.0, + 1379.0, + 946.0, + 783.0, + 946.0 + ], + "score": 1.0, + "text": "resultingfromafforestationfittedasigmoidal" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 956.0, + 1378.0, + 956.0, + 1378.0, + 979.0, + 784.0, + 979.0 + ], + "score": 1.0, + "text": "function,similartoforestgrowthfunctions.Conse-" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 990.0, + 1378.0, + 990.0, + 1378.0, + 1013.0, + 784.0, + 1013.0 + ], + "score": 1.0, + "text": "quently,weusedasigmoidalfunctiontocharacterise" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1022.0, + 1376.0, + 1022.0, + 1376.0, + 1046.0, + 784.0, + 1046.0 + ], + "score": 0.95, + "text": "the impact of plantation growth on each flow decile." + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1055.0, + 1378.0, + 1055.0, + 1378.0, + 1080.0, + 783.0, + 1080.0 + ], + "score": 0.96, + "text": "Fig.2a is a schematicof thechangein theFDCover" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1087.0, + 1137.0, + 1087.0, + 1137.0, + 1113.0, + 783.0, + 1113.0 + ], + "score": 0.97, + "text": "time. The model took the form:" + }, + { + "category_id": 15, + "poly": [ + 152.0, + 1288.0, + 712.0, + 1288.0, + 712.0, + 1313.0, + 152.0, + 1313.0 + ], + "score": 1.0, + "text": "FDCswerecomputedfromthedistributionofdaily" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1322.0, + 712.0, + 1322.0, + 712.0, + 1347.0, + 119.0, + 1347.0 + ], + "score": 0.99, + "text": "flowsforeachyearofrecordbasedonthe appropriate" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1356.0, + 714.0, + 1356.0, + 714.0, + 1378.0, + 119.0, + 1378.0 + ], + "score": 0.98, + "text": "wateryears(May-AprilorNovember-October)for" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1389.0, + 714.0, + 1389.0, + 714.0, + 1413.0, + 119.0, + 1413.0 + ], + "score": 0.97, + "text": "10Southern Hemisphere catchments.Each 10th" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1423.0, + 715.0, + 1423.0, + 715.0, + 1444.0, + 118.0, + 1444.0 + ], + "score": 0.98, + "text": "percentile(decile)wasextractedfromtheannual" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 1454.0, + 715.0, + 1454.0, + 715.0, + 1479.0, + 117.0, + 1479.0 + ], + "score": 1.0, + "text": "FDCsofeachcatchmenttoformthedatasetsfor" + }, + { + "category_id": 15, + "poly": [ + 121.0, + 1490.0, + 714.0, + 1490.0, + 714.0, + 1513.0, + 121.0, + 1513.0 + ], + "score": 0.98, + "text": "analysis.For thepurposeof characterisingchanges in" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1522.0, + 715.0, + 1522.0, + 715.0, + 1545.0, + 120.0, + 1545.0 + ], + "score": 0.99, + "text": "eachofthedeciles,itisassumedthatthetimeseriesis" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1556.0, + 714.0, + 1556.0, + 714.0, + 1579.0, + 120.0, + 1579.0 + ], + "score": 1.0, + "text": "principallyafunctionofclimateandvegetation" + }, + { + "category_id": 15, + "poly": [ + 121.0, + 1588.0, + 714.0, + 1588.0, + 714.0, + 1611.0, + 121.0, + 1611.0 + ], + "score": 1.0, + "text": "characteristics.Givenrainfallisgenerallythemost" + }, + { + "category_id": 15, + "poly": [ + 121.0, + 1622.0, + 715.0, + 1622.0, + 715.0, + 1645.0, + 121.0, + 1645.0 + ], + "score": 1.0, + "text": "importantfactoraffectingstreamflowandthemost" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1654.0, + 714.0, + 1654.0, + 714.0, + 1677.0, + 119.0, + 1677.0 + ], + "score": 1.0, + "text": "easilyaccesseddata,itischosentorepresentthe" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1687.0, + 716.0, + 1687.0, + 716.0, + 1713.0, + 119.0, + 1713.0 + ], + "score": 0.97, + "text": "climate. Catchment physical properties such as soil" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1722.0, + 714.0, + 1722.0, + 714.0, + 1744.0, + 118.0, + 1744.0 + ], + "score": 1.0, + "text": "propertiesandtopographyareassumedtobetime" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1755.0, + 715.0, + 1755.0, + 715.0, + 1778.0, + 118.0, + 1778.0 + ], + "score": 1.0, + "text": "invariantandthereforetheirimpactonrunoffis" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1786.0, + 714.0, + 1786.0, + 714.0, + 1812.0, + 118.0, + 1812.0 + ], + "score": 0.97, + "text": "considered constant throughout the analysis.As trees" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1821.0, + 712.0, + 1821.0, + 712.0, + 1847.0, + 118.0, + 1847.0 + ], + "score": 0.98, + "text": "intercept and transpire at increasing rates until canopy" + }, + { + "category_id": 15, + "poly": [ + 787.0, + 1289.0, + 856.0, + 1289.0, + 856.0, + 1313.0, + 787.0, + 1313.0 + ], + "score": 1.0, + "text": "where" + }, + { + "category_id": 15, + "poly": [ + 899.0, + 1289.0, + 1205.0, + 1289.0, + 1205.0, + 1313.0, + 899.0, + 1313.0 + ], + "score": 0.96, + "text": "is the percentile flow (i.e." + }, + { + "category_id": 15, + "poly": [ + 1253.0, + 1289.0, + 1377.0, + 1289.0, + 1377.0, + 1313.0, + 1253.0, + 1313.0 + ], + "score": 0.98, + "text": "is the 50th" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 1323.0, + 988.0, + 1323.0, + 988.0, + 1345.0, + 782.0, + 1345.0 + ], + "score": 0.98, + "text": "percentileflow)," + }, + { + "category_id": 15, + "poly": [ + 1013.0, + 1323.0, + 1076.0, + 1323.0, + 1076.0, + 1345.0, + 1013.0, + 1345.0 + ], + "score": 1.0, + "text": "and" + }, + { + "category_id": 15, + "poly": [ + 1099.0, + 1323.0, + 1378.0, + 1323.0, + 1378.0, + 1345.0, + 1099.0, + 1345.0 + ], + "score": 1.0, + "text": "arecoefficientsofthe" + }, + { + "category_id": 15, + "poly": [ + 785.0, + 1356.0, + 962.0, + 1356.0, + 962.0, + 1377.0, + 785.0, + 1377.0 + ], + "score": 0.99, + "text": "sigmoidalterm," + }, + { + "category_id": 15, + "poly": [ + 1003.0, + 1356.0, + 1379.0, + 1356.0, + 1379.0, + 1377.0, + 1003.0, + 1377.0 + ], + "score": 1.0, + "text": "isthedeviationofannualrainfall" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1388.0, + 1209.0, + 1388.0, + 1209.0, + 1413.0, + 783.0, + 1413.0 + ], + "score": 0.96, + "text": "from theperiod ofrecord average, and" + }, + { + "category_id": 15, + "poly": [ + 1263.0, + 1388.0, + 1378.0, + 1388.0, + 1378.0, + 1413.0, + 1263.0, + 1413.0 + ], + "score": 0.95, + "text": "is the time" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1422.0, + 1258.0, + 1422.0, + 1258.0, + 1446.0, + 784.0, + 1446.0 + ], + "score": 0.96, + "text": "in years at which half of the reduction in" + }, + { + "category_id": 15, + "poly": [ + 1302.0, + 1422.0, + 1379.0, + 1422.0, + 1379.0, + 1446.0, + 1302.0, + 1446.0 + ], + "score": 0.98, + "text": "due to" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1455.0, + 1378.0, + 1455.0, + 1378.0, + 1479.0, + 784.0, + 1479.0 + ], + "score": 0.98, + "text": "afforestation has taken place.For the average climate" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1487.0, + 894.0, + 1487.0, + 894.0, + 1512.0, + 783.0, + 1512.0 + ], + "score": 1.0, + "text": "condition" + }, + { + "category_id": 15, + "poly": [ + 1009.0, + 1487.0, + 1269.0, + 1487.0, + 1269.0, + 1512.0, + 1009.0, + 1512.0 + ], + "score": 0.95, + "text": "becomes the value of" + }, + { + "category_id": 15, + "poly": [ + 1312.0, + 1487.0, + 1378.0, + 1487.0, + 1378.0, + 1512.0, + 1312.0, + 1512.0 + ], + "score": 1.0, + "text": "when" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1521.0, + 1380.0, + 1521.0, + 1380.0, + 1545.0, + 783.0, + 1545.0 + ], + "score": 0.97, + "text": "the newequilibriumplantation water use under" + }, + { + "category_id": 15, + "poly": [ + 785.0, + 1555.0, + 1377.0, + 1555.0, + 1377.0, + 1578.0, + 785.0, + 1578.0 + ], + "score": 1.0, + "text": "afforestationisreached.Ythengivesthemagnitude" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1588.0, + 1238.0, + 1588.0, + 1238.0, + 1612.0, + 784.0, + 1612.0 + ], + "score": 0.95, + "text": "of change due to afforestation,and " + }, + { + "category_id": 15, + "poly": [ + 1263.0, + 1588.0, + 1377.0, + 1588.0, + 1377.0, + 1612.0, + 1263.0, + 1612.0 + ], + "score": 0.99, + "text": "describes" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 1619.0, + 1380.0, + 1619.0, + 1380.0, + 1647.0, + 782.0, + 1647.0 + ], + "score": 0.98, + "text": "the shape of the response as shown in Fig. 2b. For" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1654.0, + 1195.0, + 1654.0, + 1195.0, + 1680.0, + 783.0, + 1680.0 + ], + "score": 0.98, + "text": "the average pre-treatment condition" + }, + { + "category_id": 15, + "poly": [ + 1279.0, + 1654.0, + 1312.0, + 1654.0, + 1312.0, + 1680.0, + 1279.0, + 1680.0 + ], + "score": 1.0, + "text": "at" + }, + { + "category_id": 15, + "poly": [ + 822.0, + 1687.0, + 1074.0, + 1687.0, + 1074.0, + 1713.0, + 822.0, + 1713.0 + ], + "score": 0.99, + "text": "approximately equals" + }, + { + "category_id": 15, + "poly": [ + 1141.0, + 1687.0, + 1378.0, + 1687.0, + 1378.0, + 1713.0, + 1141.0, + 1713.0 + ], + "score": 0.98, + "text": "Estimation of a pre-" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1721.0, + 1378.0, + 1721.0, + 1378.0, + 1745.0, + 784.0, + 1745.0 + ], + "score": 0.99, + "text": "afforestationconditionwouldnot requirethetime" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 1754.0, + 1380.0, + 1754.0, + 1380.0, + 1778.0, + 782.0, + 1778.0 + ], + "score": 0.96, + "text": "term.Details of the optimisation scheme and" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1787.0, + 1378.0, + 1787.0, + 1378.0, + 1812.0, + 783.0, + 1812.0 + ], + "score": 0.96, + "text": "sensitivity tests oninitialparametervalues aregiven" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1820.0, + 1018.0, + 1820.0, + 1018.0, + 1844.0, + 783.0, + 1844.0 + ], + "score": 0.96, + "text": "in Lane et al. (2003)." + }, + { + "category_id": 15, + "poly": [ + 117.0, + 721.0, + 523.0, + 721.0, + 523.0, + 750.0, + 117.0, + 750.0 + ], + "score": 0.98, + "text": "2.1. Characterisation of fow regime" + }, + { + "category_id": 15, + "poly": [ + 1345.0, + 196.0, + 1381.0, + 196.0, + 1381.0, + 219.0, + 1345.0, + 219.0 + ], + "score": 1.0, + "text": "255" + }, + { + "category_id": 15, + "poly": [ + 468.0, + 196.0, + 1028.0, + 196.0, + 1028.0, + 220.0, + 468.0, + 220.0 + ], + "score": 0.96, + "text": "P.N.J.Lane et al./ Journal of Hydrology 310(2005) 253-265" + } + ], + "page_info": { + "page_no": 2, + "height": 2064, + "width": 1512 + } + }, + { + "layout_dets": [ + { + "category_id": 9, + "poly": [ + 1360.4927978515625, + 806.4982299804688, + 1393.562255859375, + 806.4982299804688, + 1393.562255859375, + 835.6339721679688, + 1360.4927978515625, + 835.6339721679688 + ], + "score": 0.9999969005584717 + }, + { + "category_id": 0, + "poly": [ + 794.8597412109375, + 500.950927734375, + 1061.779052734375, + 500.950927734375, + 1061.779052734375, + 529.2664794921875, + 794.8597412109375, + 529.2664794921875 + ], + "score": 0.9999960660934448 + }, + { + "category_id": 1, + "poly": [ + 795.16552734375, + 877.1219482421875, + 1396.0081787109375, + 877.1219482421875, + 1396.0081787109375, + 1240.2757568359375, + 795.16552734375, + 1240.2757568359375 + ], + "score": 0.9999922513961792 + }, + { + "category_id": 1, + "poly": [ + 795.3323974609375, + 1244.0330810546875, + 1393.7725830078125, + 1244.0330810546875, + 1393.7725830078125, + 1508.4620361328125, + 795.3323974609375, + 1508.4620361328125 + ], + "score": 0.9999896883964539 + }, + { + "category_id": 9, + "poly": [ + 1360.60400390625, + 1543.2252197265625, + 1393.3878173828125, + 1543.2252197265625, + 1393.3878173828125, + 1573.743896484375, + 1360.60400390625, + 1573.743896484375 + ], + "score": 0.99998939037323 + }, + { + "category_id": 3, + "poly": [ + 143.59278869628906, + 259.15484619140625, + 713.1118774414062, + 259.15484619140625, + 713.1118774414062, + 1178.9329833984375, + 143.59278869628906, + 1178.9329833984375 + ], + "score": 0.9999875426292419 + }, + { + "category_id": 9, + "poly": [ + 695.6785888671875, + 1697.92626953125, + 729.6533813476562, + 1697.92626953125, + 729.6533813476562, + 1727.179443359375, + 695.6785888671875, + 1727.179443359375 + ], + "score": 0.9999865293502808 + }, + { + "category_id": 1, + "poly": [ + 794.4083862304688, + 566.5299072265625, + 1394.3333740234375, + 566.5299072265625, + 1394.3333740234375, + 762.97998046875, + 794.4083862304688, + 762.97998046875 + ], + "score": 0.9999858736991882 + }, + { + "category_id": 2, + "poly": [ + 130.26800537109375, + 194.78128051757812, + 166.77423095703125, + 194.78128051757812, + 166.77423095703125, + 214.85980224609375, + 130.26800537109375, + 214.85980224609375 + ], + "score": 0.999983549118042 + }, + { + "category_id": 4, + "poly": [ + 130.88475036621094, + 1202.541748046875, + 732.7880859375, + 1202.541748046875, + 732.7880859375, + 1255.5113525390625, + 130.88475036621094, + 1255.5113525390625 + ], + "score": 0.9999827146530151 + }, + { + "category_id": 1, + "poly": [ + 131.37588500976562, + 1355.726318359375, + 730.2669067382812, + 1355.726318359375, + 730.2669067382812, + 1652.2847900390625, + 131.37588500976562, + 1652.2847900390625 + ], + "score": 0.9999791979789734 + }, + { + "category_id": 1, + "poly": [ + 131.3990020751953, + 1783.6968994140625, + 730.2479858398438, + 1783.6968994140625, + 730.2479858398438, + 1845.9527587890625, + 131.3990020751953, + 1845.9527587890625 + ], + "score": 0.9999774694442749 + }, + { + "category_id": 8, + "poly": [ + 793.2936401367188, + 779.8841552734375, + 1107.38330078125, + 779.8841552734375, + 1107.38330078125, + 863.30126953125, + 793.2936401367188, + 863.30126953125 + ], + "score": 0.9999751448631287 + }, + { + "category_id": 1, + "poly": [ + 793.5782470703125, + 254.07586669921875, + 1395.4632568359375, + 254.07586669921875, + 1395.4632568359375, + 448.5629577636719, + 793.5782470703125, + 448.5629577636719 + ], + "score": 0.9999668002128601 + }, + { + "category_id": 9, + "poly": [ + 1360.520263671875, + 1667.31787109375, + 1393.381591796875, + 1667.31787109375, + 1393.381591796875, + 1697.6356201171875, + 1360.520263671875, + 1697.6356201171875 + ], + "score": 0.9999586939811707 + }, + { + "category_id": 2, + "poly": [ + 481.1078186035156, + 195.15699768066406, + 1044.8504638671875, + 195.15699768066406, + 1044.8504638671875, + 218.15432739257812, + 481.1078186035156, + 218.15432739257812 + ], + "score": 0.9999558329582214 + }, + { + "category_id": 8, + "poly": [ + 792.6296997070312, + 1522.0426025390625, + 1110.239501953125, + 1522.0426025390625, + 1110.239501953125, + 1603.29150390625, + 792.6296997070312, + 1603.29150390625 + ], + "score": 0.9999276995658875 + }, + { + "category_id": 8, + "poly": [ + 793.1000366210938, + 1664.317138671875, + 976.23876953125, + 1664.317138671875, + 976.23876953125, + 1699.446533203125, + 793.1000366210938, + 1699.446533203125 + ], + "score": 0.9999255537986755 + }, + { + "category_id": 1, + "poly": [ + 795.8680419921875, + 1716.1470947265625, + 1394.599853515625, + 1716.1470947265625, + 1394.599853515625, + 1844.9285888671875, + 795.8680419921875, + 1844.9285888671875 + ], + "score": 0.9999062418937683 + }, + { + "category_id": 1, + "poly": [ + 792.8858642578125, + 1620.9166259765625, + 840.123046875, + 1620.9166259765625, + 840.123046875, + 1649.06201171875, + 792.8858642578125, + 1649.06201171875 + ], + "score": 0.999884307384491 + }, + { + "category_id": 8, + "poly": [ + 128.5625, + 1678.517333984375, + 566.567626953125, + 1678.517333984375, + 566.567626953125, + 1756.288330078125, + 128.5625, + 1756.288330078125 + ], + "score": 0.9992296099662781 + }, + { + "category_id": 0, + "poly": [ + 130.90809631347656, + 1288.0635986328125, + 436.2228088378906, + 1288.0635986328125, + 436.2228088378906, + 1318.1854248046875, + 130.90809631347656, + 1318.1854248046875 + ], + "score": 0.9975306987762451 + }, + { + "category_id": 14, + "poly": [ + 790, + 777, + 1108, + 777, + 1108, + 863, + 790, + 863 + ], + "score": 0.94, + "latex": "E=1.0-\\frac{\\sum_{i=1}^{N}(O_{i}-P_{i})^{2}}{\\sum_{i-1}^{N}(O_{i}-\\bar{O})^{2}}" + }, + { + "category_id": 14, + "poly": [ + 790, + 1521, + 1110, + 1521, + 1110, + 1602, + 790, + 1602 + ], + "score": 0.94, + "latex": "Q_{\\mathcal{q}_{o}}=a+\\frac{Y}{1+\\exp\\left(\\frac{T-T_{\\mathrm{half}}}{S}\\right)}" + }, + { + "category_id": 14, + "poly": [ + 125, + 1674, + 566, + 1674, + 566, + 1756, + 125, + 1756 + ], + "score": 0.93, + "latex": "N_{\\mathrm{zero}}=a+b(\\Delta P)+\\frac{Y}{1+\\exp\\left(\\frac{T-T_{\\mathrm{half}}}{S}\\right)}" + }, + { + "category_id": 13, + "poly": [ + 1306, + 319, + 1388, + 319, + 1388, + 349, + 1306, + 349 + ], + "score": 0.91, + "latex": "\\Delta P\\!=\\!0" + }, + { + "category_id": 13, + "poly": [ + 529, + 1555, + 589, + 1555, + 589, + 1585, + 529, + 1585 + ], + "score": 0.9, + "latex": "N_{\\mathrm{zero}}" + }, + { + "category_id": 13, + "poly": [ + 1281, + 1176, + 1365, + 1176, + 1365, + 1205, + 1281, + 1205 + ], + "score": 0.9, + "latex": "E\\!>\\!0.7" + }, + { + "category_id": 13, + "poly": [ + 880, + 1173, + 931, + 1173, + 931, + 1206, + 880, + 1206 + ], + "score": 0.89, + "latex": "\\!0.7" + }, + { + "category_id": 13, + "poly": [ + 160, + 1682, + 231, + 1682, + 231, + 1713, + 160, + 1713 + ], + "score": 0.88, + "latex": "a+Y)" + }, + { + "category_id": 13, + "poly": [ + 116, + 320, + 188, + 320, + 188, + 351, + 116, + 351 + ], + "score": 0.88, + "latex": "(77\\%)" + }, + { + "category_id": 13, + "poly": [ + 268, + 751, + 324, + 751, + 324, + 781, + 268, + 781 + ], + "score": 0.87, + "latex": "80\\%" + }, + { + "category_id": 13, + "poly": [ + 628, + 585, + 684, + 585, + 684, + 615, + 628, + 615 + ], + "score": 0.87, + "latex": "75\\%" + }, + { + "category_id": 13, + "poly": [ + 602, + 619, + 644, + 619, + 644, + 647, + 602, + 647 + ], + "score": 0.85, + "latex": "9\\%" + }, + { + "category_id": 13, + "poly": [ + 533, + 784, + 577, + 784, + 577, + 814, + 533, + 814 + ], + "score": 0.83, + "latex": "9\\%" + }, + { + "category_id": 13, + "poly": [ + 323, + 1384, + 364, + 1384, + 364, + 1412, + 323, + 1412 + ], + "score": 0.77, + "latex": "\\Delta P" + }, + { + "category_id": 13, + "poly": [ + 286, + 852, + 308, + 852, + 308, + 879, + 286, + 879 + ], + "score": 0.75, + "latex": "E" + }, + { + "category_id": 13, + "poly": [ + 409, + 885, + 432, + 885, + 432, + 912, + 409, + 912 + ], + "score": 0.71, + "latex": "E" + }, + { + "category_id": 13, + "poly": [ + 484, + 254, + 524, + 254, + 524, + 284, + 484, + 284 + ], + "score": 0.7, + "latex": "(E)" + }, + { + "category_id": 13, + "poly": [ + 566, + 1085, + 590, + 1085, + 590, + 1112, + 566, + 1112 + ], + "score": 0.7, + "latex": "E" + }, + { + "category_id": 13, + "poly": [ + 315, + 919, + 334, + 919, + 334, + 946, + 315, + 946 + ], + "score": 0.66, + "latex": "^b" + }, + { + "category_id": 13, + "poly": [ + 376, + 587, + 394, + 587, + 394, + 614, + 376, + 614 + ], + "score": 0.62, + "latex": "^b" + }, + { + "category_id": 13, + "poly": [ + 460, + 1051, + 478, + 1051, + 478, + 1077, + 460, + 1077 + ], + "score": 0.59, + "latex": "^b" + }, + { + "category_id": 13, + "poly": [ + 451, + 319, + 552, + 319, + 552, + 350, + 451, + 350 + ], + "score": 0.46, + "latex": "60\\%\\ 0.8" + }, + { + "category_id": 13, + "poly": [ + 498, + 719, + 522, + 719, + 522, + 746, + 498, + 746 + ], + "score": 0.45, + "latex": "Y" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 762.0, + 1379.0, + 762.0, + 1379.0, + 784.0, + 782.0, + 784.0 + ], + "score": 0.96, + "text": "Fig.3.Examples of observed and flow duration curves adjusted for" + }, + { + "category_id": 15, + "poly": [ + 781.0, + 791.0, + 1379.0, + 791.0, + 1379.0, + 810.0, + 781.0, + 810.0 + ], + "score": 0.99, + "text": "averagerainfallfollowingafforestationforStewartsCreek5," + }, + { + "category_id": 15, + "poly": [ + 782.0, + 818.0, + 868.0, + 818.0, + 868.0, + 838.0, + 782.0, + 838.0 + ], + "score": 1.0, + "text": "Australia." + }, + { + "category_id": 15, + "poly": [ + 119.0, + 258.0, + 483.0, + 258.0, + 483.0, + 282.0, + 119.0, + 282.0 + ], + "score": 0.99, + "text": "thecoefficientofefficiency" + }, + { + "category_id": 15, + "poly": [ + 525.0, + 258.0, + 714.0, + 258.0, + 714.0, + 282.0, + 525.0, + 282.0 + ], + "score": 1.0, + "text": "foreachflow" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 293.0, + 713.0, + 293.0, + 713.0, + 317.0, + 118.0, + 317.0 + ], + "score": 0.97, + "text": "percentile at all the catchments. The majority of fits" + }, + { + "category_id": 15, + "poly": [ + 189.0, + 326.0, + 295.0, + 326.0, + 295.0, + 349.0, + 189.0, + 349.0 + ], + "score": 1.0, + "text": "returned" + }, + { + "category_id": 15, + "poly": [ + 381.0, + 326.0, + 450.0, + 326.0, + 450.0, + 349.0, + 381.0, + 349.0 + ], + "score": 0.9, + "text": ",with" + }, + { + "category_id": 15, + "poly": [ + 553.0, + 326.0, + 714.0, + 326.0, + 714.0, + 349.0, + 553.0, + 349.0 + ], + "score": 0.97, + "text": "or better.The" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 358.0, + 714.0, + 358.0, + 714.0, + 382.0, + 120.0, + 382.0 + ], + "score": 0.97, + "text": "significance of the rainfall and time terms is given in" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 391.0, + 715.0, + 391.0, + 715.0, + 414.0, + 119.0, + 414.0 + ], + "score": 1.0, + "text": "Table3foralldeciles,wheresolutionswerefound." + }, + { + "category_id": 15, + "poly": [ + 119.0, + 425.0, + 714.0, + 425.0, + 714.0, + 449.0, + 119.0, + 449.0 + ], + "score": 0.96, + "text": "There were not enough data to fit the model in five" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 457.0, + 714.0, + 457.0, + 714.0, + 482.0, + 117.0, + 482.0 + ], + "score": 0.96, + "text": "instancesbecause ofextended periods of zeroflows." + }, + { + "category_id": 15, + "poly": [ + 118.0, + 490.0, + 714.0, + 490.0, + 714.0, + 516.0, + 118.0, + 516.0 + ], + "score": 0.99, + "text": "This problem is addressed to some extent in the zero" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 525.0, + 715.0, + 525.0, + 715.0, + 549.0, + 119.0, + 549.0 + ], + "score": 0.97, + "text": "flow analysis. If the rainfall signal is tobe separated" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 557.0, + 714.0, + 557.0, + 714.0, + 581.0, + 119.0, + 581.0 + ], + "score": 0.97, + "text": "from the vegetation signal the rainfall terms must be" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 592.0, + 375.0, + 592.0, + 375.0, + 616.0, + 120.0, + 616.0 + ], + "score": 0.99, + "text": "significant. This term," + }, + { + "category_id": 15, + "poly": [ + 395.0, + 592.0, + 627.0, + 592.0, + 627.0, + 616.0, + 395.0, + 616.0 + ], + "score": 0.96, + "text": ",was significant for" + }, + { + "category_id": 15, + "poly": [ + 685.0, + 592.0, + 717.0, + 592.0, + 717.0, + 616.0, + 685.0, + 616.0 + ], + "score": 1.0, + "text": "of" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 623.0, + 601.0, + 623.0, + 601.0, + 648.0, + 118.0, + 648.0 + ], + "score": 0.96, + "text": "the deciles at the 0.05 level, and a further" + }, + { + "category_id": 15, + "poly": [ + 645.0, + 623.0, + 714.0, + 623.0, + 714.0, + 648.0, + 645.0, + 648.0 + ], + "score": 1.0, + "text": "atthe" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 656.0, + 715.0, + 656.0, + 715.0, + 682.0, + 117.0, + 682.0 + ], + "score": 0.99, + "text": "0.10 level. The incidence of significance was greatest" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 691.0, + 716.0, + 691.0, + 716.0, + 714.0, + 118.0, + 714.0 + ], + "score": 1.0, + "text": "forthe10-50thpercentilesat45ofthe50datasetsat" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 724.0, + 497.0, + 724.0, + 497.0, + 747.0, + 119.0, + 747.0 + ], + "score": 0.97, + "text": "the0.051evel.Thetimeterm," + }, + { + "category_id": 15, + "poly": [ + 523.0, + 724.0, + 714.0, + 724.0, + 714.0, + 747.0, + 523.0, + 747.0 + ], + "score": 1.0, + "text": "returnedsimilar" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 757.0, + 267.0, + 757.0, + 267.0, + 781.0, + 119.0, + 781.0 + ], + "score": 0.99, + "text": "results, with" + }, + { + "category_id": 15, + "poly": [ + 325.0, + 757.0, + 714.0, + 757.0, + 714.0, + 781.0, + 325.0, + 781.0 + ], + "score": 0.94, + "text": " of the deciles significant at O.05" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 790.0, + 532.0, + 790.0, + 532.0, + 814.0, + 119.0, + 814.0 + ], + "score": 1.0, + "text": "level.Therewereanadditional" + }, + { + "category_id": 15, + "poly": [ + 578.0, + 790.0, + 714.0, + 790.0, + 714.0, + 814.0, + 578.0, + 814.0 + ], + "score": 0.97, + "text": "of deciles" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 823.0, + 431.0, + 823.0, + 431.0, + 849.0, + 120.0, + 849.0 + ], + "score": 0.99, + "text": "significant at the 0.10 level." + }, + { + "category_id": 15, + "poly": [ + 834.0, + 884.0, + 1379.0, + 884.0, + 1379.0, + 909.0, + 834.0, + 909.0 + ], + "score": 0.97, + "text": "values aregiveninTable 4.Fig.3 shows that for" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 917.0, + 1379.0, + 917.0, + 1379.0, + 941.0, + 783.0, + 941.0 + ], + "score": 0.99, + "text": "most deciles the adjusted FDCs are identical for 12" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 952.0, + 1377.0, + 952.0, + 1377.0, + 977.0, + 784.0, + 977.0 + ], + "score": 0.99, + "text": "and20yearsaftertreatment.Thisfigureclearly" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 984.0, + 1377.0, + 984.0, + 1377.0, + 1010.0, + 783.0, + 1010.0 + ], + "score": 0.96, + "text": "demonstrates the necessityfor FDC adjustment," + }, + { + "category_id": 15, + "poly": [ + 781.0, + 1017.0, + 1162.0, + 1017.0, + 1162.0, + 1043.0, + 781.0, + 1043.0 + ], + "score": 0.97, + "text": "particularly for the 20years FDC" + }, + { + "category_id": 15, + "poly": [ + 154.0, + 855.0, + 285.0, + 855.0, + 285.0, + 879.0, + 154.0, + 879.0 + ], + "score": 1.0, + "text": "Thepoorest" + }, + { + "category_id": 15, + "poly": [ + 309.0, + 855.0, + 713.0, + 855.0, + 713.0, + 879.0, + 309.0, + 879.0 + ], + "score": 0.99, + "text": "valueswere thosefromLambrechts-" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 888.0, + 408.0, + 888.0, + 408.0, + 912.0, + 117.0, + 912.0 + ], + "score": 0.98, + "text": "bos A and B. The high" + }, + { + "category_id": 15, + "poly": [ + 433.0, + 888.0, + 716.0, + 888.0, + 716.0, + 912.0, + 433.0, + 912.0 + ], + "score": 0.98, + "text": "for 50-100th deciles at" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 924.0, + 314.0, + 924.0, + 314.0, + 947.0, + 119.0, + 947.0 + ], + "score": 0.99, + "text": "Biesievlei,where" + }, + { + "category_id": 15, + "poly": [ + 335.0, + 924.0, + 714.0, + 924.0, + 714.0, + 947.0, + 335.0, + 947.0 + ], + "score": 0.96, + "text": "was notsignificant are notable.In" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 957.0, + 714.0, + 957.0, + 714.0, + 980.0, + 120.0, + 980.0 + ], + "score": 0.99, + "text": "generalthemodelfitsthehigherflows(lowerdeciles)" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 990.0, + 715.0, + 990.0, + 715.0, + 1012.0, + 119.0, + 1012.0 + ], + "score": 0.98, + "text": "better,mostofthepoorerfitsareinthe80-100" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1024.0, + 714.0, + 1024.0, + 714.0, + 1045.0, + 118.0, + 1045.0 + ], + "score": 1.0, + "text": "percentilerange.Thiscanbeexpectedgiventheresults" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1056.0, + 459.0, + 1056.0, + 459.0, + 1078.0, + 119.0, + 1078.0 + ], + "score": 0.98, + "text": "ofthesignificancetestsfor" + }, + { + "category_id": 15, + "poly": [ + 479.0, + 1056.0, + 714.0, + 1056.0, + 714.0, + 1078.0, + 479.0, + 1078.0 + ], + "score": 0.98, + "text": ".Theresultsofthe" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 1089.0, + 565.0, + 1089.0, + 565.0, + 1114.0, + 117.0, + 1114.0 + ], + "score": 0.99, + "text": "sensitivityanalysissuggestedthat the" + }, + { + "category_id": 15, + "poly": [ + 591.0, + 1089.0, + 716.0, + 1089.0, + 716.0, + 1114.0, + 591.0, + 1114.0 + ], + "score": 1.0, + "text": "valuesfor" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1122.0, + 714.0, + 1122.0, + 714.0, + 1146.0, + 119.0, + 1146.0 + ], + "score": 0.96, + "text": "Glendhu 2 and for 10th and 20th percentiles from" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1155.0, + 714.0, + 1155.0, + 714.0, + 1179.0, + 118.0, + 1179.0 + ], + "score": 0.95, + "text": "Cathedral Peak3 may exaggerate the goodness offit to" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1189.0, + 642.0, + 1189.0, + 642.0, + 1212.0, + 119.0, + 1212.0 + ], + "score": 0.97, + "text": "theexactformof themodel (Laneet al.,2003)" + }, + { + "category_id": 15, + "poly": [ + 817.0, + 1051.0, + 1379.0, + 1051.0, + 1379.0, + 1076.0, + 817.0, + 1076.0 + ], + "score": 0.97, + "text": "Therelative net flowchangedue toafforestation is" + }, + { + "category_id": 15, + "poly": [ + 785.0, + 1086.0, + 877.0, + 1086.0, + 877.0, + 1109.0, + 785.0, + 1109.0 + ], + "score": 1.0, + "text": "givenby" + }, + { + "category_id": 15, + "poly": [ + 976.0, + 1086.0, + 1377.0, + 1086.0, + 1377.0, + 1109.0, + 976.0, + 1109.0 + ], + "score": 0.99, + "text": ",whichrepresentsthechangefromthe" + }, + { + "category_id": 15, + "poly": [ + 785.0, + 1118.0, + 1379.0, + 1118.0, + 1379.0, + 1142.0, + 785.0, + 1142.0 + ], + "score": 0.96, + "text": "old equilibrium water use condition of pre-treatment" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1152.0, + 1381.0, + 1152.0, + 1381.0, + 1175.0, + 784.0, + 1175.0 + ], + "score": 1.0, + "text": "vegetationtothenewequilibriumconditionatforest" + }, + { + "category_id": 15, + "poly": [ + 785.0, + 1186.0, + 1379.0, + 1186.0, + 1379.0, + 1210.0, + 785.0, + 1210.0 + ], + "score": 0.98, + "text": "canopy closure. This quantity is plotted for all catchments" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1218.0, + 1378.0, + 1218.0, + 1378.0, + 1241.0, + 783.0, + 1241.0 + ], + "score": 0.97, + "text": "in Fig.4.Somedecileshavebeen removedfrom thedata" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1252.0, + 1378.0, + 1252.0, + 1378.0, + 1274.0, + 784.0, + 1274.0 + ], + "score": 0.97, + "text": "set,the10thand50thpercentileforGlendhu2andthe" + }, + { + "category_id": 15, + "poly": [ + 787.0, + 1284.0, + 1378.0, + 1284.0, + 1378.0, + 1308.0, + 787.0, + 1308.0 + ], + "score": 0.97, + "text": "10th and 20th percentiles from Cathedral Peak 3. The" + }, + { + "category_id": 15, + "poly": [ + 785.0, + 1317.0, + 1378.0, + 1317.0, + 1378.0, + 1342.0, + 785.0, + 1342.0 + ], + "score": 1.0, + "text": "optimisedvalueofawaszeroornearzeroforthesecases," + }, + { + "category_id": 15, + "poly": [ + 786.0, + 1352.0, + 1378.0, + 1352.0, + 1378.0, + 1375.0, + 786.0, + 1375.0 + ], + "score": 0.99, + "text": "whichisnotconsistentwith theconceptualmodel.The" + }, + { + "category_id": 15, + "poly": [ + 786.0, + 1385.0, + 1378.0, + 1385.0, + 1378.0, + 1409.0, + 786.0, + 1409.0 + ], + "score": 0.97, + "text": "changes shown in Fig. 4 are variable. However, there are" + }, + { + "category_id": 15, + "poly": [ + 785.0, + 1419.0, + 1377.0, + 1419.0, + 1377.0, + 1443.0, + 785.0, + 1443.0 + ], + "score": 0.98, + "text": "some commonalities between catchment responses. Two" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1452.0, + 1379.0, + 1452.0, + 1379.0, + 1476.0, + 783.0, + 1476.0 + ], + "score": 0.97, + "text": "types of responses (groups) were identified. Group 1" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1485.0, + 1381.0, + 1485.0, + 1381.0, + 1508.0, + 783.0, + 1508.0 + ], + "score": 0.99, + "text": "catchments showasubstantialincreaseinthenumber of" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1519.0, + 1379.0, + 1519.0, + 1379.0, + 1543.0, + 784.0, + 1543.0 + ], + "score": 0.99, + "text": "zero flow days, with a greater proportional reduction in" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1552.0, + 1380.0, + 1552.0, + 1380.0, + 1575.0, + 783.0, + 1575.0 + ], + "score": 0.99, + "text": "lowflowsthanhighflows.Group2catchmentsshowa" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 1586.0, + 1379.0, + 1586.0, + 1379.0, + 1608.0, + 782.0, + 1608.0 + ], + "score": 0.99, + "text": "moreuniformproportionalreductioninflowsacrossall" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 1618.0, + 1378.0, + 1618.0, + 1378.0, + 1643.0, + 782.0, + 1643.0 + ], + "score": 0.98, + "text": "percentiles, albeit with some variability. The catchments" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1652.0, + 974.0, + 1652.0, + 974.0, + 1676.0, + 784.0, + 1676.0 + ], + "score": 0.95, + "text": "in eachgroup are:" + }, + { + "category_id": 15, + "poly": [ + 152.0, + 1321.0, + 713.0, + 1321.0, + 713.0, + 1346.0, + 152.0, + 1346.0 + ], + "score": 0.98, + "text": "Followingthesuccessfulfittingof(2)tothe" + }, + { + "category_id": 15, + "poly": [ + 121.0, + 1356.0, + 714.0, + 1356.0, + 714.0, + 1379.0, + 121.0, + 1379.0 + ], + "score": 0.99, + "text": "observedpercentiles,theFDCswereadjustedfor" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1391.0, + 322.0, + 1391.0, + 322.0, + 1414.0, + 120.0, + 1414.0 + ], + "score": 1.0, + "text": "climatebysetting" + }, + { + "category_id": 15, + "poly": [ + 365.0, + 1391.0, + 714.0, + 1391.0, + 714.0, + 1414.0, + 365.0, + 1414.0 + ], + "score": 0.96, + "text": "to zero, representing long term" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1423.0, + 714.0, + 1423.0, + 714.0, + 1445.0, + 119.0, + 1445.0 + ], + "score": 1.0, + "text": "averageannualrainfall.TheclimateadjustedFDCs" + }, + { + "category_id": 15, + "poly": [ + 116.0, + 1454.0, + 715.0, + 1454.0, + 715.0, + 1480.0, + 116.0, + 1480.0 + ], + "score": 0.99, + "text": "produce anestimationofthechangeinflow" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 1486.0, + 716.0, + 1486.0, + 716.0, + 1513.0, + 117.0, + 1513.0 + ], + "score": 0.95, + "text": "percentiles over time for eachcatchment due to" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1520.0, + 715.0, + 1520.0, + 715.0, + 1545.0, + 119.0, + 1545.0 + ], + "score": 0.98, + "text": "afforestation thatmaybeviewed intwoforms:new" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1554.0, + 715.0, + 1554.0, + 715.0, + 1579.0, + 118.0, + 1579.0 + ], + "score": 0.97, + "text": "FDCs, adjusted for climate, as exemplified in Fig. 3" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1588.0, + 715.0, + 1588.0, + 715.0, + 1612.0, + 118.0, + 1612.0 + ], + "score": 0.97, + "text": "for Stewarts Creek 5, and a comparison between all" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1621.0, + 712.0, + 1621.0, + 712.0, + 1645.0, + 119.0, + 1645.0 + ], + "score": 0.98, + "text": "catchments of the maximum change in yield (given by" + }, + { + "category_id": 15, + "poly": [ + 121.0, + 1655.0, + 714.0, + 1655.0, + 714.0, + 1678.0, + 121.0, + 1678.0 + ], + "score": 0.98, + "text": "Y)foreachflowpercentilefrombaselineflows(given" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1688.0, + 159.0, + 1688.0, + 159.0, + 1710.0, + 119.0, + 1710.0 + ], + "score": 1.0, + "text": "by" + }, + { + "category_id": 15, + "poly": [ + 232.0, + 1688.0, + 715.0, + 1688.0, + 715.0, + 1710.0, + 232.0, + 1710.0 + ], + "score": 1.0, + "text": "asshowninFig.4.Wherethenew" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1721.0, + 714.0, + 1721.0, + 714.0, + 1744.0, + 120.0, + 1744.0 + ], + "score": 1.0, + "text": "equilibriumofmaximumwateruseisreached,the" + }, + { + "category_id": 15, + "poly": [ + 122.0, + 1755.0, + 714.0, + 1755.0, + 714.0, + 1778.0, + 122.0, + 1778.0 + ], + "score": 1.0, + "text": "adjustedFDCsforindividualyearsshouldbeidentical" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1786.0, + 713.0, + 1786.0, + 713.0, + 1811.0, + 118.0, + 1811.0 + ], + "score": 0.96, + "text": "if rainfallvariabilityhas been accounted for.The new" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1818.0, + 600.0, + 1818.0, + 600.0, + 1850.0, + 118.0, + 1850.0 + ], + "score": 0.99, + "text": "equilibrium is approximately reached for" + }, + { + "category_id": 15, + "poly": [ + 1344.0, + 196.0, + 1382.0, + 196.0, + 1382.0, + 219.0, + 1344.0, + 219.0 + ], + "score": 1.0, + "text": "259" + }, + { + "category_id": 15, + "poly": [ + 785.0, + 1712.0, + 1320.0, + 1712.0, + 1320.0, + 1736.0, + 785.0, + 1736.0 + ], + "score": 1.0, + "text": "Group 1: Stewarts Creek, Pine Creek, and Redhill" + }, + { + "category_id": 15, + "poly": [ + 785.0, + 1744.0, + 1377.0, + 1744.0, + 1377.0, + 1769.0, + 785.0, + 1769.0 + ], + "score": 0.96, + "text": "Group 2:Cathedral Peak 2 and 3,Lambrechtsbos A." + }, + { + "category_id": 15, + "poly": [ + 891.0, + 1777.0, + 1379.0, + 1777.0, + 1379.0, + 1803.0, + 891.0, + 1803.0 + ], + "score": 1.0, + "text": "Lambrechtsbos B, Glendhu 2, Biesievlei and" + }, + { + "category_id": 15, + "poly": [ + 893.0, + 1810.0, + 1067.0, + 1810.0, + 1067.0, + 1837.0, + 893.0, + 1837.0 + ], + "score": 1.0, + "text": "TraralgonCreek" + }, + { + "category_id": 15, + "poly": [ + 469.0, + 197.0, + 1029.0, + 197.0, + 1029.0, + 220.0, + 469.0, + 220.0 + ], + "score": 0.99, + "text": "P.N.J.Laneet al./JournalofHydrology310(2005)253-265" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1254.0, + 688.0, + 1254.0, + 688.0, + 1284.0, + 118.0, + 1284.0 + ], + "score": 0.97, + "text": "4.2. Adjusted FDCs—magnitude of flow reductions" + } + ], + "page_info": { + "page_no": 6, + "height": 2064, + "width": 1512 + } + }, + { + "layout_dets": [ + { + "category_id": 6, + "poly": [ + 130.8647003173828, + 1373.554931640625, + 409.2564392089844, + 1373.554931640625, + 409.2564392089844, + 1428.904541015625, + 130.8647003173828, + 1428.904541015625 + ], + "score": 0.9999980926513672 + }, + { + "category_id": 5, + "poly": [ + 125.87389373779297, + 1427.83544921875, + 1397.271484375, + 1427.83544921875, + 1397.271484375, + 1811.2467041015625, + 125.87389373779297, + 1811.2467041015625 + ], + "score": 0.999991774559021 + }, + { + "category_id": 1, + "poly": [ + 796.4015502929688, + 907.420166015625, + 1393.103759765625, + 907.420166015625, + 1393.103759765625, + 1312.5423583984375, + 796.4015502929688, + 1312.5423583984375 + ], + "score": 0.9999879598617554 + }, + { + "category_id": 1, + "poly": [ + 131.5682373046875, + 838.7388305664062, + 728.2155151367188, + 838.7388305664062, + 728.2155151367188, + 1313.8204345703125, + 131.5682373046875, + 1313.8204345703125 + ], + "score": 0.999987781047821 + }, + { + "category_id": 3, + "poly": [ + 304.8876953125, + 255.1979522705078, + 1220.4678955078125, + 255.1979522705078, + 1220.4678955078125, + 708.6326293945312, + 304.8876953125, + 708.6326293945312 + ], + "score": 0.9999865889549255 + }, + { + "category_id": 2, + "poly": [ + 131.3226318359375, + 196.32550048828125, + 166.26792907714844, + 196.32550048828125, + 166.26792907714844, + 214.9618377685547, + 131.3226318359375, + 214.9618377685547 + ], + "score": 0.9999861717224121 + }, + { + "category_id": 0, + "poly": [ + 794.9564819335938, + 839.493408203125, + 1117.9090576171875, + 839.493408203125, + 1117.9090576171875, + 868.1661987304688, + 794.9564819335938, + 868.1661987304688 + ], + "score": 0.9999852180480957 + }, + { + "category_id": 2, + "poly": [ + 481.3118591308594, + 195.47975158691406, + 1044.29833984375, + 195.47975158691406, + 1044.29833984375, + 218.9685516357422, + 481.3118591308594, + 218.9685516357422 + ], + "score": 0.9999847412109375 + }, + { + "category_id": 4, + "poly": [ + 510.6107177734375, + 733.9615478515625, + 1013.2381591796875, + 733.9615478515625, + 1013.2381591796875, + 759.1690673828125, + 510.6107177734375, + 759.1690673828125 + ], + "score": 0.9998294711112976 + }, + { + "category_id": 7, + "poly": [ + 129.30935668945312, + 1816.606689453125, + 940.408447265625, + 1816.606689453125, + 940.408447265625, + 1842.029541015625, + 129.30935668945312, + 1842.029541015625 + ], + "score": 0.9996983408927917 + }, + { + "category_id": 13, + "poly": [ + 759, + 733, + 840, + 733, + 840, + 759, + 759, + 759 + ], + "score": 0.9, + "latex": "Y/(Y\\!+\\!a)" + }, + { + "category_id": 13, + "poly": [ + 815, + 1077, + 867, + 1077, + 867, + 1108, + 815, + 1108 + ], + "score": 0.89, + "latex": "T_{\\mathrm{half}}" + }, + { + "category_id": 13, + "poly": [ + 1088, + 1179, + 1140, + 1179, + 1140, + 1211, + 1088, + 1211 + ], + "score": 0.89, + "latex": "T_{\\mathrm{half}}" + }, + { + "category_id": 13, + "poly": [ + 130, + 1247, + 196, + 1247, + 196, + 1277, + 130, + 1277 + ], + "score": 0.84, + "latex": "100\\%" + }, + { + "category_id": 13, + "poly": [ + 209, + 1042, + 276, + 1042, + 276, + 1072, + 209, + 1072 + ], + "score": 0.84, + "latex": "100\\%" + }, + { + "category_id": 13, + "poly": [ + 1174, + 940, + 1224, + 940, + 1224, + 971, + 1174, + 971 + ], + "score": 0.83, + "latex": "T_{\\mathrm{half}}" + }, + { + "category_id": 13, + "poly": [ + 129, + 1401, + 172, + 1401, + 172, + 1428, + 129, + 1428 + ], + "score": 0.7, + "latex": "T_{\\mathrm{half}}" + }, + { + "category_id": 15, + "poly": [ + 130.0, + 1375.0, + 202.0, + 1375.0, + 202.0, + 1396.0, + 130.0, + 1396.0 + ], + "score": 0.95, + "text": "Table 4" + }, + { + "category_id": 15, + "poly": [ + 173.0, + 1404.0, + 406.0, + 1404.0, + 406.0, + 1426.0, + 173.0, + 1426.0 + ], + "score": 0.98, + "text": "(years) for all catchments" + }, + { + "category_id": 15, + "poly": [ + 831.0, + 911.0, + 1391.0, + 911.0, + 1391.0, + 935.0, + 831.0, + 935.0 + ], + "score": 0.97, + "text": "The speed of flow responses to afforestation can be" + }, + { + "category_id": 15, + "poly": [ + 795.0, + 944.0, + 1173.0, + 944.0, + 1173.0, + 971.0, + 795.0, + 971.0 + ], + "score": 0.96, + "text": "evaluated by examining the value of" + }, + { + "category_id": 15, + "poly": [ + 1225.0, + 944.0, + 1393.0, + 944.0, + 1393.0, + 971.0, + 1225.0, + 971.0 + ], + "score": 0.98, + "text": "(Table 4).There" + }, + { + "category_id": 15, + "poly": [ + 794.0, + 977.0, + 1393.0, + 977.0, + 1393.0, + 1005.0, + 794.0, + 1005.0 + ], + "score": 1.0, + "text": "is substantial variation in response times both over the" + }, + { + "category_id": 15, + "poly": [ + 795.0, + 1015.0, + 1392.0, + 1015.0, + 1392.0, + 1036.0, + 795.0, + 1036.0 + ], + "score": 1.0, + "text": "percentilespreadinsomeindividualcatchments,and" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1047.0, + 1392.0, + 1047.0, + 1392.0, + 1071.0, + 796.0, + 1071.0 + ], + "score": 0.96, + "text": "between the catchments. The majority of responses have" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1083.0, + 814.0, + 1083.0, + 814.0, + 1105.0, + 796.0, + 1105.0 + ], + "score": 0.96, + "text": "a" + }, + { + "category_id": 15, + "poly": [ + 868.0, + 1083.0, + 1392.0, + 1083.0, + 1392.0, + 1105.0, + 868.0, + 1105.0 + ], + "score": 0.99, + "text": "valuebetween5and10years.PineCreekand" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1116.0, + 1392.0, + 1116.0, + 1392.0, + 1139.0, + 796.0, + 1139.0 + ], + "score": 0.99, + "text": "StewartsCreek,Redhill andLambrechtsbosAexhibitthe" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1150.0, + 1393.0, + 1150.0, + 1393.0, + 1175.0, + 796.0, + 1175.0 + ], + "score": 1.0, + "text": "fastestresponses,withBiesievleishowingthemost" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1183.0, + 1087.0, + 1183.0, + 1087.0, + 1209.0, + 796.0, + 1209.0 + ], + "score": 0.98, + "text": "uniformly slowresponse." + }, + { + "category_id": 15, + "poly": [ + 1141.0, + 1183.0, + 1393.0, + 1183.0, + 1393.0, + 1209.0, + 1141.0, + 1209.0 + ], + "score": 0.99, + "text": "for theSouthAfrican" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 1219.0, + 1392.0, + 1219.0, + 1392.0, + 1243.0, + 797.0, + 1243.0 + ], + "score": 0.98, + "text": "catchments display a good correspondence to published" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 1253.0, + 1392.0, + 1253.0, + 1392.0, + 1276.0, + 797.0, + 1276.0 + ], + "score": 0.98, + "text": "annualchanges(Scottetal.,2000;VanWyk,1987)," + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1288.0, + 1392.0, + 1288.0, + 1392.0, + 1310.0, + 796.0, + 1310.0 + ], + "score": 0.99, + "text": "exceptingthe10-20thdecilesforbothCathedralPeak" + }, + { + "category_id": 15, + "poly": [ + 166.0, + 843.0, + 729.0, + 843.0, + 729.0, + 867.0, + 166.0, + 867.0 + ], + "score": 0.94, + "text": "Group1 exhibit both the highest reduction of" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 875.0, + 728.0, + 875.0, + 728.0, + 902.0, + 131.0, + 902.0 + ], + "score": 0.97, + "text": "flows overall, and show the largest proportional" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 910.0, + 728.0, + 910.0, + 728.0, + 939.0, + 131.0, + 939.0 + ], + "score": 0.97, + "text": "reduction at lower flows, leading to a complete" + }, + { + "category_id": 15, + "poly": [ + 130.0, + 946.0, + 729.0, + 946.0, + 729.0, + 969.0, + 130.0, + 969.0 + ], + "score": 0.99, + "text": "cessationofflow.Comparisonof flowreductionsis" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 980.0, + 729.0, + 980.0, + 729.0, + 1003.0, + 131.0, + 1003.0 + ], + "score": 1.0, + "text": "hinderedslightlybytherangeofafforestationatthe" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1014.0, + 729.0, + 1014.0, + 729.0, + 1037.0, + 131.0, + 1037.0 + ], + "score": 0.97, + "text": "catchments(Table 1).These results couldbe scaled" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1048.0, + 208.0, + 1048.0, + 208.0, + 1072.0, + 131.0, + 1072.0 + ], + "score": 0.99, + "text": "upto" + }, + { + "category_id": 15, + "poly": [ + 277.0, + 1048.0, + 731.0, + 1048.0, + 731.0, + 1072.0, + 277.0, + 1072.0 + ], + "score": 0.93, + "text": " afforested if it is assumed there is a" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1082.0, + 726.0, + 1082.0, + 726.0, + 1105.0, + 131.0, + 1105.0 + ], + "score": 0.97, + "text": "linear relationshipbetweentheareaplantedandflow" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1116.0, + 728.0, + 1116.0, + 728.0, + 1139.0, + 132.0, + 1139.0 + ], + "score": 0.99, + "text": "reductions.Asthereisnoevidencethatthisisthe" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1151.0, + 728.0, + 1151.0, + 728.0, + 1175.0, + 132.0, + 1175.0 + ], + "score": 0.96, + "text": "case we have not presented scaled reductions here." + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1184.0, + 730.0, + 1184.0, + 730.0, + 1209.0, + 131.0, + 1209.0 + ], + "score": 0.98, + "text": "Linear scaling wouldshiftthereductioncurves" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1218.0, + 728.0, + 1218.0, + 728.0, + 1242.0, + 132.0, + 1242.0 + ], + "score": 0.94, + "text": "upward for those catchments that are less than" + }, + { + "category_id": 15, + "poly": [ + 197.0, + 1250.0, + 728.0, + 1250.0, + 728.0, + 1279.0, + 197.0, + 1279.0 + ], + "score": 0.95, + "text": " afforested, but would not change the shape" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1287.0, + 485.0, + 1287.0, + 485.0, + 1314.0, + 132.0, + 1314.0 + ], + "score": 0.98, + "text": "of the curves or our groupings." + }, + { + "category_id": 15, + "poly": [ + 130.0, + 196.0, + 167.0, + 196.0, + 167.0, + 218.0, + 130.0, + 218.0 + ], + "score": 1.0, + "text": "260" + }, + { + "category_id": 15, + "poly": [ + 794.0, + 840.0, + 1118.0, + 840.0, + 1118.0, + 869.0, + 794.0, + 869.0 + ], + "score": 0.99, + "text": "4.3. Timing of fow reductions" + }, + { + "category_id": 15, + "poly": [ + 481.0, + 196.0, + 1042.0, + 196.0, + 1042.0, + 220.0, + 481.0, + 220.0 + ], + "score": 0.96, + "text": "P.N.J.Lane et al./ Journal of Hydrology 310(2005) 253-265" + }, + { + "category_id": 15, + "poly": [ + 513.0, + 737.0, + 758.0, + 737.0, + 758.0, + 760.0, + 513.0, + 760.0 + ], + "score": 0.95, + "text": "Fig. 4. Net flow reductions" + }, + { + "category_id": 15, + "poly": [ + 841.0, + 737.0, + 1009.0, + 737.0, + 1009.0, + 760.0, + 841.0, + 760.0 + ], + "score": 0.99, + "text": "for all catchments" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1820.0, + 936.0, + 1820.0, + 936.0, + 1843.0, + 131.0, + 1843.0 + ], + "score": 0.97, + "text": "Note that no solution could be found for the 5O percentile for Glendhu indicted by the ns" + } + ], + "page_info": { + "page_no": 7, + "height": 2064, + "width": 1512 + } + }, + { + "layout_dets": [ + { + "category_id": 6, + "poly": [ + 114.02448272705078, + 252.96278381347656, + 1384.26953125, + 252.96278381347656, + 1384.26953125, + 336.2478332519531, + 114.02448272705078, + 336.2478332519531 + ], + "score": 0.9999914169311523 + }, + { + "category_id": 1, + "poly": [ + 782.927978515625, + 1059.080810546875, + 1379.0045166015625, + 1059.080810546875, + 1379.0045166015625, + 1518.4134521484375, + 782.927978515625, + 1518.4134521484375 + ], + "score": 0.9999841451644897 + }, + { + "category_id": 1, + "poly": [ + 117.70939636230469, + 1242.48876953125, + 715.529052734375, + 1242.48876953125, + 715.529052734375, + 1847.1575927734375, + 117.70939636230469, + 1847.1575927734375 + ], + "score": 0.9999837279319763 + }, + { + "category_id": 5, + "poly": [ + 112.87651062011719, + 340.12884521484375, + 1386.6334228515625, + 340.12884521484375, + 1386.6334228515625, + 690.6670532226562, + 112.87651062011719, + 690.6670532226562 + ], + "score": 0.9999768733978271 + }, + { + "category_id": 0, + "poly": [ + 782.4056396484375, + 1617.4853515625, + 928.7228393554688, + 1617.4853515625, + 928.7228393554688, + 1646.1173095703125, + 782.4056396484375, + 1646.1173095703125 + ], + "score": 0.9999604225158691 + }, + { + "category_id": 1, + "poly": [ + 782.9052734375, + 1683.71533203125, + 1379.5103759765625, + 1683.71533203125, + 1379.5103759765625, + 1845.6593017578125, + 782.9052734375, + 1845.6593017578125 + ], + "score": 0.9999464750289917 + }, + { + "category_id": 1, + "poly": [ + 119.27631378173828, + 994.8147583007812, + 712.2225341796875, + 994.8147583007812, + 712.2225341796875, + 1120.854736328125, + 119.27631378173828, + 1120.854736328125 + ], + "score": 0.9999319911003113 + }, + { + "category_id": 0, + "poly": [ + 783.0419921875, + 993.3599243164062, + 987.883056640625, + 993.3599243164062, + 987.883056640625, + 1020.2771606445312, + 783.0419921875, + 1020.2771606445312 + ], + "score": 0.9998669624328613 + }, + { + "category_id": 2, + "poly": [ + 1346.875, + 196.13433837890625, + 1379.426025390625, + 196.13433837890625, + 1379.426025390625, + 215.6661834716797, + 1346.875, + 215.6661834716797 + ], + "score": 0.9998455047607422 + }, + { + "category_id": 0, + "poly": [ + 119.71259307861328, + 1178.468994140625, + 627.7359008789062, + 1178.468994140625, + 627.7359008789062, + 1205.4873046875, + 119.71259307861328, + 1205.4873046875 + ], + "score": 0.9995098114013672 + }, + { + "category_id": 2, + "poly": [ + 464.848388671875, + 193.9952392578125, + 1033.1187744140625, + 193.9952392578125, + 1033.1187744140625, + 218.6742706298828, + 464.848388671875, + 218.6742706298828 + ], + "score": 0.9992669820785522 + }, + { + "category_id": 7, + "poly": [ + 116.6382827758789, + 699.0242309570312, + 1381.2982177734375, + 699.0242309570312, + 1381.2982177734375, + 856.2595825195312, + 116.6382827758789, + 856.2595825195312 + ], + "score": 0.9891747236251831 + }, + { + "category_id": 13, + "poly": [ + 458, + 778, + 601, + 778, + 601, + 806, + 458, + 806 + ], + "score": 0.91, + "latex": "\\sum Y/\\sum(a+Y)" + }, + { + "category_id": 13, + "poly": [ + 169, + 1025, + 221, + 1025, + 221, + 1056, + 169, + 1056 + ], + "score": 0.91, + "latex": "T_{\\mathrm{half}}" + }, + { + "category_id": 13, + "poly": [ + 464, + 750, + 607, + 750, + 607, + 778, + 464, + 778 + ], + "score": 0.88, + "latex": "\\sum Y/\\sum(a+Y)" + }, + { + "category_id": 13, + "poly": [ + 1201, + 1191, + 1277, + 1191, + 1277, + 1221, + 1201, + 1221 + ], + "score": 0.88, + "latex": "\\Delta N_{\\mathrm{zero}}" + }, + { + "category_id": 13, + "poly": [ + 1296, + 1323, + 1350, + 1323, + 1350, + 1353, + 1296, + 1353 + ], + "score": 0.86, + "latex": "50\\%" + }, + { + "category_id": 13, + "poly": [ + 1078, + 1159, + 1101, + 1159, + 1101, + 1185, + 1078, + 1185 + ], + "score": 0.77, + "latex": "E" + }, + { + "category_id": 13, + "poly": [ + 1113, + 1192, + 1133, + 1192, + 1133, + 1219, + 1113, + 1219 + ], + "score": 0.69, + "latex": "^b" + }, + { + "category_id": 13, + "poly": [ + 375, + 811, + 390, + 811, + 390, + 830, + 375, + 830 + ], + "score": 0.67, + "latex": "a" + }, + { + "category_id": 13, + "poly": [ + 990, + 1196, + 1003, + 1196, + 1003, + 1218, + 990, + 1218 + ], + "score": 0.61, + "latex": "t^{\\star}" + }, + { + "category_id": 13, + "poly": [ + 1066, + 812, + 1080, + 812, + 1080, + 830, + 1066, + 830 + ], + "score": 0.58, + "latex": "a" + }, + { + "category_id": 13, + "poly": [ + 431, + 808, + 448, + 808, + 448, + 830, + 431, + 830 + ], + "score": 0.46, + "latex": "Y" + }, + { + "category_id": 13, + "poly": [ + 1246, + 1357, + 1283, + 1357, + 1283, + 1386, + 1246, + 1386 + ], + "score": 0.43, + "latex": "\\mathrm{Ck}" + }, + { + "category_id": 13, + "poly": [ + 773, + 779, + 827, + 779, + 827, + 804, + 773, + 804 + ], + "score": 0.42, + "latex": "100\\mathrm{th}" + }, + { + "category_id": 13, + "poly": [ + 1107, + 1357, + 1144, + 1357, + 1144, + 1386, + 1107, + 1386 + ], + "score": 0.41, + "latex": "\\mathrm{Ck}" + }, + { + "category_id": 13, + "poly": [ + 640, + 807, + 684, + 807, + 684, + 831, + 640, + 831 + ], + "score": 0.29, + "latex": "20\\mathrm{th}" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 257.0, + 188.0, + 257.0, + 188.0, + 277.0, + 117.0, + 277.0 + ], + "score": 1.0, + "text": "Table5" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 284.0, + 1381.0, + 284.0, + 1381.0, + 308.0, + 117.0, + 308.0 + ], + "score": 0.99, + "text": "Published flow reductions from paired catchment analyses, after Scott et al. (2000), Hickel (2001), Nandakumar and Mein (1993) and Fahey and" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 313.0, + 679.0, + 313.0, + 679.0, + 337.0, + 117.0, + 337.0 + ], + "score": 0.96, + "text": "Jackson(1997) compared to estimated reductions in this study" + }, + { + "category_id": 15, + "poly": [ + 818.0, + 1063.0, + 1379.0, + 1063.0, + 1379.0, + 1087.0, + 818.0, + 1087.0 + ], + "score": 0.98, + "text": "As this analysis could only be applied, where there" + }, + { + "category_id": 15, + "poly": [ + 781.0, + 1095.0, + 1380.0, + 1095.0, + 1380.0, + 1122.0, + 781.0, + 1122.0 + ], + "score": 0.98, + "text": "was consistent drying up of streams, it was confined to" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1130.0, + 1378.0, + 1130.0, + 1378.0, + 1151.0, + 784.0, + 1151.0 + ], + "score": 0.99, + "text": "StewartsCreek,PineCreekandRedhillcatchments.The" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1162.0, + 1077.0, + 1162.0, + 1077.0, + 1186.0, + 783.0, + 1186.0 + ], + "score": 1.0, + "text": "modelreturnedvaluesof" + }, + { + "category_id": 15, + "poly": [ + 1102.0, + 1162.0, + 1378.0, + 1162.0, + 1378.0, + 1186.0, + 1102.0, + 1186.0 + ], + "score": 0.94, + "text": "of 0.95, 0.99and 0.99," + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1197.0, + 989.0, + 1197.0, + 989.0, + 1222.0, + 784.0, + 1222.0 + ], + "score": 1.0, + "text": "respectively.The" + }, + { + "category_id": 15, + "poly": [ + 1004.0, + 1197.0, + 1112.0, + 1197.0, + 1112.0, + 1222.0, + 1004.0, + 1222.0 + ], + "score": 0.98, + "text": "-testson" + }, + { + "category_id": 15, + "poly": [ + 1134.0, + 1197.0, + 1200.0, + 1197.0, + 1200.0, + 1222.0, + 1134.0, + 1222.0 + ], + "score": 1.0, + "text": "and" + }, + { + "category_id": 15, + "poly": [ + 1278.0, + 1197.0, + 1380.0, + 1197.0, + 1380.0, + 1222.0, + 1278.0, + 1222.0 + ], + "score": 1.0, + "text": "returned" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1230.0, + 1380.0, + 1230.0, + 1380.0, + 1254.0, + 783.0, + 1254.0 + ], + "score": 0.97, + "text": "significant results at the 0.05 level for both parameters at" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1261.0, + 1379.0, + 1261.0, + 1379.0, + 1286.0, + 783.0, + 1286.0 + ], + "score": 1.0, + "text": "allthreecatchments.Theclimateadjustedzeroflow" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1295.0, + 1377.0, + 1295.0, + 1377.0, + 1319.0, + 783.0, + 1319.0 + ], + "score": 0.98, + "text": "days are shown in Fig. 5. The increases in zero flow days" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 1330.0, + 1295.0, + 1330.0, + 1295.0, + 1351.0, + 782.0, + 1351.0 + ], + "score": 1.0, + "text": "aresubstantialwithflowsconfinedtolessthan" + }, + { + "category_id": 15, + "poly": [ + 1351.0, + 1330.0, + 1379.0, + 1330.0, + 1379.0, + 1351.0, + 1351.0, + 1351.0 + ], + "score": 1.0, + "text": "of" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1360.0, + 1106.0, + 1360.0, + 1106.0, + 1387.0, + 783.0, + 1387.0 + ], + "score": 0.99, + "text": "thetimebyyear8 atStewarts" + }, + { + "category_id": 15, + "poly": [ + 1145.0, + 1360.0, + 1245.0, + 1360.0, + 1245.0, + 1387.0, + 1145.0, + 1387.0 + ], + "score": 0.96, + "text": "and Pine" + }, + { + "category_id": 15, + "poly": [ + 1284.0, + 1360.0, + 1378.0, + 1360.0, + 1378.0, + 1387.0, + 1284.0, + 1387.0 + ], + "score": 0.99, + "text": "and year" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1393.0, + 1380.0, + 1393.0, + 1380.0, + 1420.0, + 783.0, + 1420.0 + ], + "score": 0.99, + "text": "11 at Redhill. The latter has changed from an almost" + }, + { + "category_id": 15, + "poly": [ + 779.0, + 1428.0, + 1379.0, + 1428.0, + 1379.0, + 1454.0, + 779.0, + 1454.0 + ], + "score": 0.98, + "text": "permanent to a highly intermittent stream. The curves" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 1462.0, + 1379.0, + 1462.0, + 1379.0, + 1486.0, + 782.0, + 1486.0 + ], + "score": 0.96, + "text": "are also in sensible agreement with the flow reductions" + }, + { + "category_id": 15, + "poly": [ + 781.0, + 1493.0, + 882.0, + 1493.0, + 882.0, + 1521.0, + 781.0, + 1521.0 + ], + "score": 1.0, + "text": "in Fig. 4." + }, + { + "category_id": 15, + "poly": [ + 152.0, + 1247.0, + 714.0, + 1247.0, + 714.0, + 1272.0, + 152.0, + 1272.0 + ], + "score": 0.94, + "text": "A further check on the overall model performance is" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1283.0, + 714.0, + 1283.0, + 714.0, + 1304.0, + 119.0, + 1304.0 + ], + "score": 1.0, + "text": "acomparisonwithpublishedresultsofpairedcatchment" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1316.0, + 714.0, + 1316.0, + 714.0, + 1339.0, + 120.0, + 1339.0 + ], + "score": 1.0, + "text": "studies.Thedatathatcanbecomparedwithourresults" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1350.0, + 714.0, + 1350.0, + 714.0, + 1374.0, + 120.0, + 1374.0 + ], + "score": 0.97, + "text": "are presented in Table 5 and can be broadly compared" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1381.0, + 714.0, + 1381.0, + 714.0, + 1406.0, + 119.0, + 1406.0 + ], + "score": 0.97, + "text": "with Fig. 4. These data are reductions in years with near" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1415.0, + 717.0, + 1415.0, + 717.0, + 1440.0, + 118.0, + 1440.0 + ], + "score": 0.98, + "text": "average annual rainfall,and at a time after treatment" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1450.0, + 714.0, + 1450.0, + 714.0, + 1474.0, + 120.0, + 1474.0 + ], + "score": 0.98, + "text": "when maximum changes in streamflow have occurred." + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1483.0, + 714.0, + 1483.0, + 714.0, + 1506.0, + 119.0, + 1506.0 + ], + "score": 1.0, + "text": "Table5alsoincludesestimatesonthetotalandlowflow" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1516.0, + 713.0, + 1516.0, + 713.0, + 1541.0, + 118.0, + 1541.0 + ], + "score": 0.97, + "text": "reductions calculated from this study.Results from Pine" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1552.0, + 716.0, + 1552.0, + 716.0, + 1576.0, + 120.0, + 1576.0 + ], + "score": 0.98, + "text": "Creek and Traralgon Creek are not included in Table 5" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1584.0, + 715.0, + 1584.0, + 715.0, + 1609.0, + 118.0, + 1609.0 + ], + "score": 0.97, + "text": "as these catchments are not paired.Exact comparisons" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1619.0, + 714.0, + 1619.0, + 714.0, + 1643.0, + 120.0, + 1643.0 + ], + "score": 0.98, + "text": "are impossible because of the rainfall variability, and" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 1650.0, + 713.0, + 1650.0, + 713.0, + 1678.0, + 117.0, + 1678.0 + ], + "score": 0.96, + "text": "lack of calibration period for Redhill. Despite this," + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1684.0, + 714.0, + 1684.0, + 714.0, + 1709.0, + 118.0, + 1709.0 + ], + "score": 1.0, + "text": "Table5showsthattotalandlowflowreductions" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1720.0, + 714.0, + 1720.0, + 714.0, + 1744.0, + 120.0, + 1744.0 + ], + "score": 0.98, + "text": "estimated from our study are comparable to the results" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1753.0, + 714.0, + 1753.0, + 714.0, + 1776.0, + 119.0, + 1776.0 + ], + "score": 0.98, + "text": "frompairedcatchmentstudies,indicating that our" + }, + { + "category_id": 15, + "poly": [ + 121.0, + 1787.0, + 714.0, + 1787.0, + 714.0, + 1811.0, + 121.0, + 1811.0 + ], + "score": 0.95, + "text": "simple model has successfully removed the rainfall" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1820.0, + 187.0, + 1820.0, + 187.0, + 1846.0, + 120.0, + 1846.0 + ], + "score": 0.95, + "text": "signal." + }, + { + "category_id": 15, + "poly": [ + 781.0, + 1617.0, + 928.0, + 1617.0, + 928.0, + 1645.0, + 781.0, + 1645.0 + ], + "score": 1.0, + "text": "5. Discussion" + }, + { + "category_id": 15, + "poly": [ + 817.0, + 1687.0, + 1378.0, + 1687.0, + 1378.0, + 1712.0, + 817.0, + 1712.0 + ], + "score": 0.97, + "text": "The aims of the project have largely been met. The" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 1721.0, + 1377.0, + 1721.0, + 1377.0, + 1745.0, + 782.0, + 1745.0 + ], + "score": 0.97, + "text": "general characterisation of FDCs and adjustment fon" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 1754.0, + 1380.0, + 1754.0, + 1380.0, + 1781.0, + 782.0, + 1781.0 + ], + "score": 0.99, + "text": "climate has been very encouraging given the task of" + }, + { + "category_id": 15, + "poly": [ + 785.0, + 1788.0, + 1380.0, + 1788.0, + 1380.0, + 1812.0, + 785.0, + 1812.0 + ], + "score": 0.96, + "text": "fitting our model to 10 flow percentiles, for 10 different" + }, + { + "category_id": 15, + "poly": [ + 781.0, + 1819.0, + 1378.0, + 1819.0, + 1378.0, + 1846.0, + 781.0, + 1846.0 + ], + "score": 0.93, + "text": "catchments (resulting in 1oo model fits) with" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 995.0, + 711.0, + 995.0, + 711.0, + 1019.0, + 118.0, + 1019.0 + ], + "score": 0.98, + "text": "catchments and the lower deciles at Lambrechtsbos B" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 1028.0, + 168.0, + 1028.0, + 168.0, + 1056.0, + 117.0, + 1056.0 + ], + "score": 1.0, + "text": "The" + }, + { + "category_id": 15, + "poly": [ + 222.0, + 1028.0, + 710.0, + 1028.0, + 710.0, + 1056.0, + 222.0, + 1056.0 + ], + "score": 0.97, + "text": "from Glendhu 2 appears to be substantially" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1064.0, + 711.0, + 1064.0, + 711.0, + 1088.0, + 118.0, + 1088.0 + ], + "score": 0.98, + "text": "lower than other published data (Fahey and Jackson" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1095.0, + 187.0, + 1095.0, + 187.0, + 1122.0, + 118.0, + 1122.0 + ], + "score": 0.99, + "text": "1997)." + }, + { + "category_id": 15, + "poly": [ + 783.0, + 994.0, + 987.0, + 994.0, + 987.0, + 1020.0, + 783.0, + 1020.0 + ], + "score": 1.0, + "text": "4.5.Zerofowdays" + }, + { + "category_id": 15, + "poly": [ + 1345.0, + 195.0, + 1381.0, + 195.0, + 1381.0, + 219.0, + 1345.0, + 219.0 + ], + "score": 1.0, + "text": "261" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1182.0, + 626.0, + 1182.0, + 626.0, + 1205.0, + 120.0, + 1205.0 + ], + "score": 0.99, + "text": "4.4.Comparisonwithpairedcatchmentstudies" + }, + { + "category_id": 15, + "poly": [ + 469.0, + 197.0, + 1028.0, + 197.0, + 1028.0, + 220.0, + 469.0, + 220.0 + ], + "score": 0.99, + "text": "P.N.J.Laneet al./JournalofHydrology310(2005)253-265" + }, + { + "category_id": 15, + "poly": [ + 127.0, + 696.0, + 1381.0, + 696.0, + 1381.0, + 727.0, + 127.0, + 727.0 + ], + "score": 0.99, + "text": "a Rainfall refers to the rainfall in the year used for comparison of results. The value in brackets refers to the deviation from the mean annual" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 728.0, + 406.0, + 728.0, + 406.0, + 751.0, + 117.0, + 751.0 + ], + "score": 0.97, + "text": "rainfall for the period of record." + }, + { + "category_id": 15, + "poly": [ + 124.0, + 748.0, + 463.0, + 748.0, + 463.0, + 782.0, + 124.0, + 782.0 + ], + "score": 0.99, + "text": "b Total fow reduction calculated by" + }, + { + "category_id": 15, + "poly": [ + 608.0, + 748.0, + 740.0, + 748.0, + 740.0, + 782.0, + 608.0, + 782.0 + ], + "score": 1.0, + "text": "for all deciles." + }, + { + "category_id": 15, + "poly": [ + 126.0, + 777.0, + 457.0, + 777.0, + 457.0, + 808.0, + 126.0, + 808.0 + ], + "score": 0.97, + "text": "c Low fow reduction calculated by" + }, + { + "category_id": 15, + "poly": [ + 602.0, + 777.0, + 772.0, + 777.0, + 772.0, + 808.0, + 602.0, + 808.0 + ], + "score": 0.99, + "text": "for 70, 80, 90 and" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 777.0, + 933.0, + 777.0, + 933.0, + 808.0, + 828.0, + 808.0 + ], + "score": 0.97, + "text": "percentiles." + }, + { + "category_id": 15, + "poly": [ + 126.0, + 803.0, + 374.0, + 803.0, + 374.0, + 839.0, + 126.0, + 839.0 + ], + "score": 0.97, + "text": "d For Cathedral Peak 3 the" + }, + { + "category_id": 15, + "poly": [ + 391.0, + 803.0, + 430.0, + 803.0, + 430.0, + 839.0, + 391.0, + 839.0 + ], + "score": 1.0, + "text": "and" + }, + { + "category_id": 15, + "poly": [ + 449.0, + 803.0, + 639.0, + 803.0, + 639.0, + 839.0, + 449.0, + 839.0 + ], + "score": 1.0, + "text": "values for the 10 and" + }, + { + "category_id": 15, + "poly": [ + 685.0, + 803.0, + 1065.0, + 803.0, + 1065.0, + 839.0, + 685.0, + 839.0 + ], + "score": 0.99, + "text": "percentiles were excluded as the values of" + }, + { + "category_id": 15, + "poly": [ + 1081.0, + 803.0, + 1380.0, + 803.0, + 1380.0, + 839.0, + 1081.0, + 839.0 + ], + "score": 0.99, + "text": "were lower then the values of the" + }, + { + "category_id": 15, + "poly": [ + 116.0, + 836.0, + 309.0, + 836.0, + 309.0, + 861.0, + 116.0, + 861.0 + ], + "score": 0.98, + "text": "30-100th percentiles." + } + ], + "page_info": { + "page_no": 8, + "height": 2064, + "width": 1512 + } + }, + { + "layout_dets": [ + { + "category_id": 4, + "poly": [ + 131.2040557861328, + 1337.5989990234375, + 733.35986328125, + 1337.5989990234375, + 733.35986328125, + 1417.8536376953125, + 131.2040557861328, + 1417.8536376953125 + ], + "score": 0.9999973177909851 + }, + { + "category_id": 1, + "poly": [ + 131.22463989257812, + 1444.04736328125, + 731.0685424804688, + 1444.04736328125, + 731.0685424804688, + 1847.82861328125, + 131.22463989257812, + 1847.82861328125 + ], + "score": 0.9999927282333374 + }, + { + "category_id": 1, + "poly": [ + 794.774169921875, + 255.46112060546875, + 1396.0106201171875, + 255.46112060546875, + 1396.0106201171875, + 1082.3992919921875, + 794.774169921875, + 1082.3992919921875 + ], + "score": 0.9999898672103882 + }, + { + "category_id": 3, + "poly": [ + 134.4781036376953, + 257.3968200683594, + 736.1229248046875, + 257.3968200683594, + 736.1229248046875, + 1314.7186279296875, + 134.4781036376953, + 1314.7186279296875 + ], + "score": 0.9999856948852539 + }, + { + "category_id": 2, + "poly": [ + 131.43182373046875, + 196.0782012939453, + 165.26609802246094, + 196.0782012939453, + 165.26609802246094, + 215.07960510253906, + 131.43182373046875, + 215.07960510253906 + ], + "score": 0.9999679923057556 + }, + { + "category_id": 1, + "poly": [ + 795.1036376953125, + 1085.7911376953125, + 1395.2923583984375, + 1085.7911376953125, + 1395.2923583984375, + 1847.192138671875, + 795.1036376953125, + 1847.192138671875 + ], + "score": 0.9999593496322632 + }, + { + "category_id": 2, + "poly": [ + 480.6620178222656, + 195.42437744140625, + 1044.6075439453125, + 195.42437744140625, + 1044.6075439453125, + 218.66488647460938, + 480.6620178222656, + 218.66488647460938 + ], + "score": 0.9999507665634155 + }, + { + "category_id": 13, + "poly": [ + 1045, + 452, + 1098, + 452, + 1098, + 482, + 1045, + 482 + ], + "score": 0.87, + "latex": "27\\%" + }, + { + "category_id": 15, + "poly": [ + 130.0, + 1341.0, + 729.0, + 1341.0, + 729.0, + 1364.0, + 130.0, + 1364.0 + ], + "score": 0.96, + "text": "Fig.5. Number of zero flow days for average rainfall following" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1369.0, + 727.0, + 1369.0, + 727.0, + 1388.0, + 132.0, + 1388.0 + ], + "score": 0.99, + "text": "afforestationforStewartsCreek5,RedhillandPineCreek." + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1397.0, + 218.0, + 1397.0, + 218.0, + 1416.0, + 131.0, + 1416.0 + ], + "score": 0.96, + "text": "Australia." + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1447.0, + 727.0, + 1447.0, + 727.0, + 1476.0, + 132.0, + 1476.0 + ], + "score": 0.99, + "text": "substantially varying spatial scales, soils and geology," + }, + { + "category_id": 15, + "poly": [ + 133.0, + 1483.0, + 727.0, + 1483.0, + 727.0, + 1506.0, + 133.0, + 1506.0 + ], + "score": 1.0, + "text": "speciesplantedandclimaticenvironments.Although" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1516.0, + 727.0, + 1516.0, + 727.0, + 1539.0, + 132.0, + 1539.0 + ], + "score": 0.96, + "text": "there were poor resultsfor individual deciles,the FDCs" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 1551.0, + 728.0, + 1551.0, + 728.0, + 1575.0, + 133.0, + 1575.0 + ], + "score": 0.96, + "text": "at eight of the10 catchments were adequatelydescribed" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1583.0, + 729.0, + 1583.0, + 729.0, + 1607.0, + 131.0, + 1607.0 + ], + "score": 0.99, + "text": "by Eq. (2). The results of the statistical tests in which the" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1618.0, + 727.0, + 1618.0, + 727.0, + 1642.0, + 132.0, + 1642.0 + ], + "score": 0.98, + "text": "rainfall term was significant for most deciles demon-" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 1651.0, + 727.0, + 1651.0, + 727.0, + 1678.0, + 133.0, + 1678.0 + ], + "score": 0.98, + "text": "strated the model structure was appropriate for adjusting" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1685.0, + 728.0, + 1685.0, + 728.0, + 1709.0, + 132.0, + 1709.0 + ], + "score": 0.95, + "text": "the FDCs for climatic (rainfall) variability.The" + }, + { + "category_id": 15, + "poly": [ + 134.0, + 1720.0, + 728.0, + 1720.0, + 728.0, + 1744.0, + 134.0, + 1744.0 + ], + "score": 0.96, + "text": "comparisons of our results with published paired" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1752.0, + 728.0, + 1752.0, + 728.0, + 1777.0, + 131.0, + 1777.0 + ], + "score": 0.97, + "text": "catchmentanalyses are satisfactory,althoughthe" + }, + { + "category_id": 15, + "poly": [ + 134.0, + 1787.0, + 729.0, + 1787.0, + 729.0, + 1811.0, + 134.0, + 1811.0 + ], + "score": 0.97, + "text": "different methodologies make direct comparisons of" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 1820.0, + 730.0, + 1820.0, + 730.0, + 1844.0, + 133.0, + 1844.0 + ], + "score": 0.98, + "text": "deciles withtotalflowuncertain.Lowflows at" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 259.0, + 1390.0, + 259.0, + 1390.0, + 282.0, + 796.0, + 282.0 + ], + "score": 1.0, + "text": "LambrechtsbosBappeartobeover-estimatedbyour" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 294.0, + 1390.0, + 294.0, + 1390.0, + 317.0, + 797.0, + 317.0 + ], + "score": 0.96, + "text": "model, which is unsurprising as the model fit was poor." + }, + { + "category_id": 15, + "poly": [ + 798.0, + 326.0, + 1392.0, + 326.0, + 1392.0, + 350.0, + 798.0, + 350.0 + ], + "score": 0.99, + "text": "The remaining four South African catchments, and also" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 359.0, + 1392.0, + 359.0, + 1392.0, + 383.0, + 796.0, + 383.0 + ], + "score": 0.97, + "text": "Redhill and Stewarts Creek are in good agreement with" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 391.0, + 1395.0, + 391.0, + 1395.0, + 418.0, + 797.0, + 418.0 + ], + "score": 1.0, + "text": "the published values, particularly when the deviation of" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 426.0, + 1392.0, + 426.0, + 1392.0, + 450.0, + 797.0, + 450.0 + ], + "score": 0.97, + "text": "average rainfall is considered. Glendhu 2 reductions are" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 459.0, + 1044.0, + 459.0, + 1044.0, + 483.0, + 797.0, + 483.0 + ], + "score": 1.0, + "text": "closetothereported" + }, + { + "category_id": 15, + "poly": [ + 1099.0, + 459.0, + 1392.0, + 459.0, + 1392.0, + 483.0, + 1099.0, + 483.0 + ], + "score": 0.95, + "text": ",but our model produces" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 493.0, + 1390.0, + 493.0, + 1390.0, + 516.0, + 796.0, + 516.0 + ], + "score": 0.95, + "text": "a heavier impact on thelower flows. Overall, it appears" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 525.0, + 1393.0, + 525.0, + 1393.0, + 548.0, + 796.0, + 548.0 + ], + "score": 1.0, + "text": "therearenosignificantdiscrepancieswiththepublished" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 559.0, + 1391.0, + 559.0, + 1391.0, + 583.0, + 796.0, + 583.0 + ], + "score": 0.98, + "text": "paired catchment analyses.We suggest our technique" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 593.0, + 1393.0, + 593.0, + 1393.0, + 615.0, + 796.0, + 615.0 + ], + "score": 1.0, + "text": "representsanalternativetothepaired-catchmentmethod" + }, + { + "category_id": 15, + "poly": [ + 793.0, + 623.0, + 1392.0, + 623.0, + 1392.0, + 651.0, + 793.0, + 651.0 + ], + "score": 0.98, + "text": "for assessing hydrologic response to vegetation treat-" + }, + { + "category_id": 15, + "poly": [ + 795.0, + 659.0, + 1393.0, + 659.0, + 1393.0, + 679.0, + 795.0, + 679.0 + ], + "score": 1.0, + "text": "ment,wherepaireddataareunavailable.Themethod" + }, + { + "category_id": 15, + "poly": [ + 794.0, + 690.0, + 1392.0, + 690.0, + 1392.0, + 713.0, + 794.0, + 713.0 + ], + "score": 0.95, + "text": "has not yet resultedin a predictive model,but has" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 725.0, + 1392.0, + 725.0, + 1392.0, + 748.0, + 796.0, + 748.0 + ], + "score": 0.98, + "text": "increasedourknowledgeof afforestationimpacts.This" + }, + { + "category_id": 15, + "poly": [ + 794.0, + 756.0, + 1396.0, + 756.0, + 1396.0, + 783.0, + 794.0, + 783.0 + ], + "score": 0.98, + "text": "is a valuable outcome given the contentious issue of" + }, + { + "category_id": 15, + "poly": [ + 798.0, + 789.0, + 1392.0, + 789.0, + 1392.0, + 813.0, + 798.0, + 813.0 + ], + "score": 0.97, + "text": "afforestation in Australia and other countries, and a" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 824.0, + 1393.0, + 824.0, + 1393.0, + 846.0, + 796.0, + 846.0 + ], + "score": 1.0, + "text": "currentpaucityofdataoninter-annualflows.Itshould" + }, + { + "category_id": 15, + "poly": [ + 793.0, + 855.0, + 1391.0, + 855.0, + 1391.0, + 884.0, + 793.0, + 884.0 + ], + "score": 0.98, + "text": "be noted that nine of the 10 catchment were pine species." + }, + { + "category_id": 15, + "poly": [ + 794.0, + 889.0, + 1394.0, + 889.0, + 1394.0, + 914.0, + 794.0, + 914.0 + ], + "score": 1.0, + "text": "Moredataisrequiredtocomparetheimpactof" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 924.0, + 1391.0, + 924.0, + 1391.0, + 948.0, + 796.0, + 948.0 + ], + "score": 0.99, + "text": "hardwood species, particularly eucalypts, on the FDC." + }, + { + "category_id": 15, + "poly": [ + 798.0, + 956.0, + 1391.0, + 956.0, + 1391.0, + 979.0, + 798.0, + 979.0 + ], + "score": 0.98, + "text": "Unfortunately thesedata arecurrentlyscarce.Thereare" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 988.0, + 1392.0, + 988.0, + 1392.0, + 1014.0, + 796.0, + 1014.0 + ], + "score": 0.97, + "text": "substantial data on the physiological controls of eucalypt" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 1024.0, + 1393.0, + 1024.0, + 1393.0, + 1047.0, + 797.0, + 1047.0 + ], + "score": 0.95, + "text": "water use(seeWhitehead andBeadle,2004),butnot at" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 1058.0, + 1010.0, + 1058.0, + 1010.0, + 1079.0, + 797.0, + 1079.0 + ], + "score": 1.0, + "text": "thecatchmentscale" + }, + { + "category_id": 15, + "poly": [ + 129.0, + 195.0, + 167.0, + 195.0, + 167.0, + 219.0, + 129.0, + 219.0 + ], + "score": 1.0, + "text": "262" + }, + { + "category_id": 15, + "poly": [ + 830.0, + 1088.0, + 1393.0, + 1088.0, + 1393.0, + 1113.0, + 830.0, + 1113.0 + ], + "score": 0.99, + "text": "Themodelfitsshowwehavequantifiedthenet" + }, + { + "category_id": 15, + "poly": [ + 798.0, + 1123.0, + 1390.0, + 1123.0, + 1390.0, + 1147.0, + 798.0, + 1147.0 + ], + "score": 0.99, + "text": "impactofafforestationforthemajorityoftheflow" + }, + { + "category_id": 15, + "poly": [ + 794.0, + 1156.0, + 1393.0, + 1156.0, + 1393.0, + 1180.0, + 794.0, + 1180.0 + ], + "score": 0.99, + "text": "percentiles in the 10 catchments. Results for the 10-50th" + }, + { + "category_id": 15, + "poly": [ + 794.0, + 1190.0, + 1394.0, + 1190.0, + 1394.0, + 1215.0, + 794.0, + 1215.0 + ], + "score": 0.99, + "text": "percentiles wereparticularlyencouraging.Itisnot" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1224.0, + 1391.0, + 1224.0, + 1391.0, + 1245.0, + 796.0, + 1245.0 + ], + "score": 0.99, + "text": "surprisingthattherelationshipbetweenrainfallandflow" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1254.0, + 1391.0, + 1254.0, + 1391.0, + 1279.0, + 796.0, + 1279.0 + ], + "score": 0.98, + "text": "diminishes at lower flows (60-100th percentile),where" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1289.0, + 1392.0, + 1289.0, + 1392.0, + 1312.0, + 796.0, + 1312.0 + ], + "score": 1.0, + "text": "seasonalstorageeffectsandrainfalldistributionbecome" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1323.0, + 1393.0, + 1323.0, + 1393.0, + 1345.0, + 796.0, + 1345.0 + ], + "score": 0.99, + "text": "moreimportantdriversforrunoffgeneration.The" + }, + { + "category_id": 15, + "poly": [ + 794.0, + 1355.0, + 1393.0, + 1355.0, + 1393.0, + 1378.0, + 794.0, + 1378.0 + ], + "score": 0.99, + "text": "poorestmodelfitsweregainedforLambrechtsbosA" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 1388.0, + 1392.0, + 1388.0, + 1392.0, + 1411.0, + 797.0, + 1411.0 + ], + "score": 0.99, + "text": "andB.ThelikelyreasonatLambrechtsbosAisan" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 1421.0, + 1393.0, + 1421.0, + 1393.0, + 1444.0, + 797.0, + 1444.0 + ], + "score": 1.0, + "text": "observedannualdecreaseinstandwateruseafter12" + }, + { + "category_id": 15, + "poly": [ + 795.0, + 1456.0, + 1392.0, + 1456.0, + 1392.0, + 1477.0, + 795.0, + 1477.0 + ], + "score": 0.98, + "text": "years(Scottetal.,2000)whichdoesnotconformtothe" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 1489.0, + 1394.0, + 1489.0, + 1394.0, + 1511.0, + 797.0, + 1511.0 + ], + "score": 0.99, + "text": "sigmoidalformofourmodeloverthefull19yearsof" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1521.0, + 1393.0, + 1521.0, + 1393.0, + 1544.0, + 796.0, + 1544.0 + ], + "score": 0.97, + "text": "record.Thefailureof themodel tofit thelower flows at" + }, + { + "category_id": 15, + "poly": [ + 794.0, + 1553.0, + 1393.0, + 1553.0, + 1393.0, + 1578.0, + 794.0, + 1578.0 + ], + "score": 0.97, + "text": "LambrechtsbosB is not as explicable.A decrease in" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 1588.0, + 1392.0, + 1588.0, + 1392.0, + 1611.0, + 797.0, + 1611.0 + ], + "score": 1.0, + "text": "standwateruseinthiscatchmentisobservedasthe" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1622.0, + 1392.0, + 1622.0, + 1392.0, + 1645.0, + 796.0, + 1645.0 + ], + "score": 0.96, + "text": "plantation ages,butdoesnot occur during thefirst 20" + }, + { + "category_id": 15, + "poly": [ + 795.0, + 1656.0, + 1391.0, + 1656.0, + 1391.0, + 1677.0, + 795.0, + 1677.0 + ], + "score": 0.98, + "text": "yearsaftertreatment(Scottetal.,2000).Otherdatafrom" + }, + { + "category_id": 15, + "poly": [ + 796.0, + 1685.0, + 1392.0, + 1685.0, + 1392.0, + 1711.0, + 796.0, + 1711.0 + ], + "score": 0.92, + "text": "South Africa (Scott et al.,2000)indicate there are" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 1718.0, + 1389.0, + 1718.0, + 1389.0, + 1745.0, + 797.0, + 1745.0 + ], + "score": 0.98, + "text": "diminished flow reductions as plantations age, but again" + }, + { + "category_id": 15, + "poly": [ + 795.0, + 1754.0, + 1393.0, + 1754.0, + 1393.0, + 1779.0, + 795.0, + 1779.0 + ], + "score": 0.97, + "text": "generally after 20 years.Our use of an asymptotic curve" + }, + { + "category_id": 15, + "poly": [ + 797.0, + 1788.0, + 1393.0, + 1788.0, + 1393.0, + 1810.0, + 797.0, + 1810.0 + ], + "score": 1.0, + "text": "assumesanewequilibriumofstandwateruseis" + }, + { + "category_id": 15, + "poly": [ + 795.0, + 1818.0, + 1390.0, + 1818.0, + 1390.0, + 1846.0, + 795.0, + 1846.0 + ], + "score": 0.98, + "text": "reached. The results of the model fitting generally justify" + }, + { + "category_id": 15, + "poly": [ + 481.0, + 196.0, + 1042.0, + 196.0, + 1042.0, + 220.0, + 481.0, + 220.0 + ], + "score": 0.96, + "text": "P.N.J.Lane et al./ Journal of Hydrology 310(2005) 253-265" + } + ], + "page_info": { + "page_no": 9, + "height": 2064, + "width": 1512 + } + }, + { + "layout_dets": [ + { + "category_id": 1, + "poly": [ + 117.55418395996094, + 251.62586975097656, + 717.4569702148438, + 251.62586975097656, + 717.4569702148438, + 582.5938720703125, + 117.55418395996094, + 582.5938720703125 + ], + "score": 0.9999960660934448 + }, + { + "category_id": 1, + "poly": [ + 781.7598266601562, + 253.2869873046875, + 1382.113525390625, + 253.2869873046875, + 1382.113525390625, + 749.5455322265625, + 781.7598266601562, + 749.5455322265625 + ], + "score": 0.9999920129776001 + }, + { + "category_id": 1, + "poly": [ + 780.9730224609375, + 1416.1875, + 1381.69287109375, + 1416.1875, + 1381.69287109375, + 1847.7166748046875, + 780.9730224609375, + 1847.7166748046875 + ], + "score": 0.9999914169311523 + }, + { + "category_id": 1, + "poly": [ + 117.01396942138672, + 1416.33447265625, + 717.9161987304688, + 1416.33447265625, + 717.9161987304688, + 1848.1839599609375, + 117.01396942138672, + 1848.1839599609375 + ], + "score": 0.9999870657920837 + }, + { + "category_id": 1, + "poly": [ + 781.1908569335938, + 752.2108154296875, + 1380.9827880859375, + 752.2108154296875, + 1380.9827880859375, + 1280.6949462890625, + 781.1908569335938, + 1280.6949462890625 + ], + "score": 0.9999850988388062 + }, + { + "category_id": 1, + "poly": [ + 117.43836975097656, + 885.7890014648438, + 718.2291870117188, + 885.7890014648438, + 718.2291870117188, + 1415.08984375, + 117.43836975097656, + 1415.08984375 + ], + "score": 0.9999825358390808 + }, + { + "category_id": 1, + "poly": [ + 118.48710632324219, + 587.1766357421875, + 716.4562377929688, + 587.1766357421875, + 716.4562377929688, + 883.3397216796875, + 118.48710632324219, + 883.3397216796875 + ], + "score": 0.9999718070030212 + }, + { + "category_id": 2, + "poly": [ + 1346.4129638671875, + 196.0000762939453, + 1381.5977783203125, + 196.0000762939453, + 1381.5977783203125, + 216.04489135742188, + 1346.4129638671875, + 216.04489135742188 + ], + "score": 0.9999490976333618 + }, + { + "category_id": 0, + "poly": [ + 781.2360229492188, + 1351.0572509765625, + 1112.8250732421875, + 1351.0572509765625, + 1112.8250732421875, + 1380.2672119140625, + 781.2360229492188, + 1380.2672119140625 + ], + "score": 0.9996971487998962 + }, + { + "category_id": 2, + "poly": [ + 466.37176513671875, + 195.2890167236328, + 1031.220458984375, + 195.2890167236328, + 1031.220458984375, + 219.7666778564453, + 466.37176513671875, + 219.7666778564453 + ], + "score": 0.9995570182800293 + }, + { + "category_id": 13, + "poly": [ + 510, + 1017, + 563, + 1017, + 563, + 1047, + 510, + 1047 + ], + "score": 0.89, + "latex": "85\\%" + }, + { + "category_id": 13, + "poly": [ + 1121, + 321, + 1143, + 321, + 1143, + 347, + 1121, + 347 + ], + "score": 0.55, + "latex": "E." + }, + { + "category_id": 13, + "poly": [ + 433, + 354, + 456, + 354, + 456, + 380, + 433, + 380 + ], + "score": 0.46, + "latex": "E." + }, + { + "category_id": 13, + "poly": [ + 578, + 1018, + 683, + 1018, + 683, + 1048, + 578, + 1048 + ], + "score": 0.39, + "latex": "1260\\,\\mathrm{mm}" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 257.0, + 713.0, + 257.0, + 713.0, + 283.0, + 120.0, + 283.0 + ], + "score": 0.98, + "text": "this assumption for the length of commercial plantation" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 293.0, + 711.0, + 293.0, + 711.0, + 317.0, + 119.0, + 317.0 + ], + "score": 0.97, + "text": "growth (up to 20 years) considered here. The physio-" + }, + { + "category_id": 15, + "poly": [ + 121.0, + 325.0, + 714.0, + 325.0, + 714.0, + 349.0, + 121.0, + 349.0 + ], + "score": 0.97, + "text": "logical relationshipbetween stand age andwaterusefor" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 360.0, + 432.0, + 360.0, + 432.0, + 383.0, + 119.0, + 383.0 + ], + "score": 1.0, + "text": "plantationspeciesotherthan" + }, + { + "category_id": 15, + "poly": [ + 457.0, + 360.0, + 715.0, + 360.0, + 715.0, + 383.0, + 457.0, + 383.0 + ], + "score": 1.0, + "text": "regnanshavenotbeen" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 388.0, + 714.0, + 388.0, + 714.0, + 419.0, + 118.0, + 419.0 + ], + "score": 0.99, + "text": "thoroughly investigated, although Cornish and Vertessy" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 422.0, + 716.0, + 422.0, + 716.0, + 454.0, + 117.0, + 454.0 + ], + "score": 0.98, + "text": "(2001) and Roberts et al. (2001) have shown young" + }, + { + "category_id": 15, + "poly": [ + 121.0, + 461.0, + 714.0, + 461.0, + 714.0, + 482.0, + 121.0, + 482.0 + ], + "score": 1.0, + "text": "mixedspecieseucalyptforestsmayusemorewaterthan" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 490.0, + 714.0, + 490.0, + 714.0, + 517.0, + 118.0, + 517.0 + ], + "score": 0.99, + "text": "mature stands, and Putahena and Cordery (2000) suggest" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 525.0, + 714.0, + 525.0, + 714.0, + 549.0, + 120.0, + 549.0 + ], + "score": 1.0, + "text": "maximumPinusradiatawaterusemayhavebeen" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 559.0, + 654.0, + 559.0, + 654.0, + 580.0, + 120.0, + 580.0 + ], + "score": 0.99, + "text": "reachedafter12years,withasubsequentdecline." + }, + { + "category_id": 15, + "poly": [ + 819.0, + 259.0, + 1376.0, + 259.0, + 1376.0, + 282.0, + 819.0, + 282.0 + ], + "score": 0.99, + "text": "TraralgonCreekwouldbeexpected tohaveboththe" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 292.0, + 1377.0, + 292.0, + 1377.0, + 317.0, + 782.0, + 317.0 + ], + "score": 0.97, + "text": "most subdued flow reductions and longer response time" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 325.0, + 1120.0, + 325.0, + 1120.0, + 351.0, + 782.0, + 351.0 + ], + "score": 0.93, + "text": "because of the large area of" + }, + { + "category_id": 15, + "poly": [ + 1144.0, + 325.0, + 1381.0, + 325.0, + 1381.0, + 351.0, + 1144.0, + 351.0 + ], + "score": 0.98, + "text": "regnans forest,and" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 359.0, + 1380.0, + 359.0, + 1380.0, + 380.0, + 784.0, + 380.0 + ], + "score": 1.0, + "text": "uncertainvegetationrecord.Peakstandwateruseofa" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 389.0, + 1377.0, + 389.0, + 1377.0, + 417.0, + 782.0, + 417.0 + ], + "score": 0.96, + "text": "natural stand of this species is around 30 years." + }, + { + "category_id": 15, + "poly": [ + 786.0, + 425.0, + 1377.0, + 425.0, + 1377.0, + 448.0, + 786.0, + 448.0 + ], + "score": 0.98, + "text": "Additionallyinthislarge,‘realworld'catchment" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 458.0, + 1379.0, + 458.0, + 1379.0, + 483.0, + 783.0, + 483.0 + ], + "score": 1.0, + "text": "thereisacontinuouscycleofforestmanagement" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 489.0, + 1380.0, + 489.0, + 1380.0, + 517.0, + 783.0, + 517.0 + ], + "score": 0.99, + "text": "which includes harvesting. A mixture of pasture and" + }, + { + "category_id": 15, + "poly": [ + 785.0, + 522.0, + 1379.0, + 522.0, + 1379.0, + 551.0, + 785.0, + 551.0 + ], + "score": 0.98, + "text": "'scrub', which could represent significant understorey" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 558.0, + 1377.0, + 558.0, + 1377.0, + 580.0, + 784.0, + 580.0 + ], + "score": 0.99, + "text": "stands,werereplacedbyplantationspecies.Conse-" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 591.0, + 1382.0, + 591.0, + 1382.0, + 617.0, + 783.0, + 617.0 + ], + "score": 0.99, + "text": "quently the difference between pre and post treatment" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 625.0, + 1381.0, + 625.0, + 1381.0, + 648.0, + 783.0, + 648.0 + ], + "score": 0.96, + "text": "ETmaybeless than atother catchments.Reductions of" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 657.0, + 1377.0, + 657.0, + 1377.0, + 682.0, + 784.0, + 682.0 + ], + "score": 0.96, + "text": "this magnitude could be more readily expected in larger," + }, + { + "category_id": 15, + "poly": [ + 783.0, + 689.0, + 1378.0, + 689.0, + 1378.0, + 715.0, + 783.0, + 715.0 + ], + "score": 0.97, + "text": "multi land use catchments than the very high impacts" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 723.0, + 1291.0, + 723.0, + 1291.0, + 747.0, + 784.0, + 747.0 + ], + "score": 0.98, + "text": "estimated at the smaller Australian catchments." + }, + { + "category_id": 15, + "poly": [ + 818.0, + 1421.0, + 1380.0, + 1421.0, + 1380.0, + 1449.0, + 818.0, + 1449.0 + ], + "score": 0.97, + "text": "This project sought to (i) develop a method to remove" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1455.0, + 1377.0, + 1455.0, + 1377.0, + 1479.0, + 783.0, + 1479.0 + ], + "score": 0.96, + "text": "the climate signal from streamflow records to identify" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1489.0, + 1379.0, + 1489.0, + 1379.0, + 1512.0, + 784.0, + 1512.0 + ], + "score": 0.99, + "text": "theimpactofvegetationonflowfromafforested" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1522.0, + 1378.0, + 1522.0, + 1378.0, + 1546.0, + 784.0, + 1546.0 + ], + "score": 0.97, + "text": "catchments,and (i) quantify this impact on the flow" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1554.0, + 1380.0, + 1554.0, + 1380.0, + 1581.0, + 783.0, + 1581.0 + ], + "score": 0.96, + "text": "duration curve. A simple model was proposed that" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 1588.0, + 1380.0, + 1588.0, + 1380.0, + 1613.0, + 782.0, + 1613.0 + ], + "score": 0.97, + "text": "considered the age of plantation and the annual rainfall" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1623.0, + 1378.0, + 1623.0, + 1378.0, + 1646.0, + 784.0, + 1646.0 + ], + "score": 0.98, + "text": "tobetheprincipal driversfor evapotranspiration.This" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1655.0, + 1378.0, + 1655.0, + 1378.0, + 1677.0, + 784.0, + 1677.0 + ], + "score": 0.99, + "text": "modelwasfittedtotheobserveddecilesoftheFDC,and" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1688.0, + 1378.0, + 1688.0, + 1378.0, + 1712.0, + 784.0, + 1712.0 + ], + "score": 0.97, + "text": "the climate signal was then removed from the stream-" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 1718.0, + 1380.0, + 1718.0, + 1380.0, + 1746.0, + 782.0, + 1746.0 + ], + "score": 0.98, + "text": "flow records by adjusting the FDC for average rainfall" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1755.0, + 1378.0, + 1755.0, + 1378.0, + 1777.0, + 783.0, + 1777.0 + ], + "score": 1.0, + "text": "overtheperiodofrecord.Themodelwastestedand" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1787.0, + 1377.0, + 1787.0, + 1377.0, + 1811.0, + 784.0, + 1811.0 + ], + "score": 0.97, + "text": "applied to 10 afforested catchments.We successfully" + }, + { + "category_id": 15, + "poly": [ + 782.0, + 1818.0, + 1379.0, + 1818.0, + 1379.0, + 1845.0, + 782.0, + 1845.0 + ], + "score": 0.96, + "text": "fitted our model to catchments with varying spatial" + }, + { + "category_id": 15, + "poly": [ + 154.0, + 1422.0, + 713.0, + 1422.0, + 713.0, + 1448.0, + 154.0, + 1448.0 + ], + "score": 0.98, + "text": "The response groups may be in part explained by the" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1457.0, + 714.0, + 1457.0, + 714.0, + 1478.0, + 119.0, + 1478.0 + ], + "score": 1.0, + "text": "storagecharacteristicsofthecatchments.Accurate" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1489.0, + 714.0, + 1489.0, + 714.0, + 1512.0, + 118.0, + 1512.0 + ], + "score": 0.99, + "text": "measuresofstoragearenotavailablefrom theliterature," + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1521.0, + 715.0, + 1521.0, + 715.0, + 1544.0, + 119.0, + 1544.0 + ], + "score": 0.97, + "text": "but thesoil depths and thebaseflowindex(Table1)both" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1555.0, + 714.0, + 1555.0, + 714.0, + 1577.0, + 120.0, + 1577.0 + ], + "score": 1.0, + "text": "showthethreesoutheasternAustraliancatchmentswith" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1589.0, + 715.0, + 1589.0, + 715.0, + 1612.0, + 120.0, + 1612.0 + ], + "score": 1.0, + "text": "thegreatestreductionarelikelytohavethelowest" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1622.0, + 712.0, + 1622.0, + 712.0, + 1646.0, + 120.0, + 1646.0 + ], + "score": 0.99, + "text": "storage capacity. The greater flow reductions, particu-" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1653.0, + 714.0, + 1653.0, + 714.0, + 1679.0, + 119.0, + 1679.0 + ], + "score": 0.95, + "text": "larly for low flows,could be expected under these" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1687.0, + 714.0, + 1687.0, + 714.0, + 1711.0, + 118.0, + 1711.0 + ], + "score": 0.99, + "text": "conditions. Inclusion of a storage term in the model is an" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1719.0, + 715.0, + 1719.0, + 715.0, + 1746.0, + 120.0, + 1746.0 + ], + "score": 0.99, + "text": "obvious option for improving the analysis. However the" + }, + { + "category_id": 15, + "poly": [ + 121.0, + 1755.0, + 716.0, + 1755.0, + 716.0, + 1779.0, + 121.0, + 1779.0 + ], + "score": 0.96, + "text": "addition of extra parameters would be at the cost of" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1789.0, + 712.0, + 1789.0, + 712.0, + 1812.0, + 120.0, + 1812.0 + ], + "score": 0.97, + "text": "maintaining model simplicity,particularly as character" + }, + { + "category_id": 15, + "poly": [ + 120.0, + 1821.0, + 517.0, + 1821.0, + 517.0, + 1847.0, + 120.0, + 1847.0 + ], + "score": 0.99, + "text": "ising a transient storage is not trivial." + }, + { + "category_id": 15, + "poly": [ + 818.0, + 758.0, + 1376.0, + 758.0, + 1376.0, + 781.0, + 818.0, + 781.0 + ], + "score": 1.0, + "text": "Theanalysisofzeroflowdayswassuccessful" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 790.0, + 1378.0, + 790.0, + 1378.0, + 815.0, + 784.0, + 815.0 + ], + "score": 0.98, + "text": "demonstrating that theimpact onflowintermittence can" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 824.0, + 1377.0, + 824.0, + 1377.0, + 847.0, + 783.0, + 847.0 + ], + "score": 0.99, + "text": "beevaluatedwithoutof theentireFDC.Thiswashelpful" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 857.0, + 1379.0, + 857.0, + 1379.0, + 881.0, + 784.0, + 881.0 + ], + "score": 0.99, + "text": "as the change in the higher percentiles (low flows) could" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 890.0, + 1378.0, + 890.0, + 1378.0, + 912.0, + 783.0, + 912.0 + ], + "score": 1.0, + "text": "notalwaysbemodelled.Theresultsforthethree" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 924.0, + 1378.0, + 924.0, + 1378.0, + 945.0, + 784.0, + 945.0 + ], + "score": 1.0, + "text": "catchmentsanalysedarearatherstarkindicationofthe" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 958.0, + 1378.0, + 958.0, + 1378.0, + 980.0, + 783.0, + 980.0 + ], + "score": 1.0, + "text": "potentialforhighlyincreasedzeroflowperiodsinsmall" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 989.0, + 1377.0, + 989.0, + 1377.0, + 1013.0, + 783.0, + 1013.0 + ], + "score": 0.98, + "text": "catchments, at least in south-eastern Australia. However." + }, + { + "category_id": 15, + "poly": [ + 781.0, + 1022.0, + 1381.0, + 1022.0, + 1381.0, + 1046.0, + 781.0, + 1046.0 + ], + "score": 0.99, + "text": "it should be noted these curves probably represent a" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1056.0, + 1379.0, + 1056.0, + 1379.0, + 1079.0, + 783.0, + 1079.0 + ], + "score": 1.0, + "text": "maximumresponseastheyareallderivedfromsmall" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1090.0, + 1376.0, + 1090.0, + 1376.0, + 1114.0, + 784.0, + 1114.0 + ], + "score": 0.96, + "text": "catchments with smallstorage capacities and large" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1123.0, + 1379.0, + 1123.0, + 1379.0, + 1146.0, + 783.0, + 1146.0 + ], + "score": 0.99, + "text": "percentagesof afforestation.Thismethodcouldbeused" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1156.0, + 1377.0, + 1156.0, + 1377.0, + 1180.0, + 783.0, + 1180.0 + ], + "score": 0.95, + "text": "to determine change in the occurrence of any given flow" + }, + { + "category_id": 15, + "poly": [ + 781.0, + 1188.0, + 1379.0, + 1188.0, + 1379.0, + 1213.0, + 781.0, + 1213.0 + ], + "score": 1.0, + "text": "inresponsetoafforestation;e.g.todeterminethe" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1219.0, + 1379.0, + 1219.0, + 1379.0, + 1249.0, + 783.0, + 1249.0 + ], + "score": 0.94, + "text": "likelihood of maintaining a reservoir storage or an" + }, + { + "category_id": 15, + "poly": [ + 783.0, + 1255.0, + 1376.0, + 1255.0, + 1376.0, + 1280.0, + 783.0, + 1280.0 + ], + "score": 0.98, + "text": "environmentalflowthatrequires anaveragecriticalflow." + }, + { + "category_id": 15, + "poly": [ + 154.0, + 889.0, + 714.0, + 889.0, + 714.0, + 914.0, + 154.0, + 914.0 + ], + "score": 0.99, + "text": "Themagnitudeof theresponsewithinGroup2varies" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 924.0, + 715.0, + 924.0, + 715.0, + 946.0, + 118.0, + 946.0 + ], + "score": 0.99, + "text": "considerably,withgreaterreductioninflowsinthetwo" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 955.0, + 714.0, + 955.0, + 714.0, + 980.0, + 118.0, + 980.0 + ], + "score": 0.98, + "text": "Cathedral Peak catchments,and Lambrechtsbos B." + }, + { + "category_id": 15, + "poly": [ + 119.0, + 991.0, + 714.0, + 991.0, + 714.0, + 1014.0, + 119.0, + 1014.0 + ], + "score": 1.0, + "text": "Potentialevaporationisinphasewithrainfallatthe" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1024.0, + 509.0, + 1024.0, + 509.0, + 1047.0, + 119.0, + 1047.0 + ], + "score": 1.0, + "text": "CathedralPeaksitesastheyreceive" + }, + { + "category_id": 15, + "poly": [ + 684.0, + 1024.0, + 715.0, + 1024.0, + 715.0, + 1047.0, + 684.0, + 1047.0 + ], + "score": 0.81, + "text": "on" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1056.0, + 716.0, + 1056.0, + 716.0, + 1079.0, + 119.0, + 1079.0 + ], + "score": 0.98, + "text": "average)of theirrainfallinsummer.Theconjunction of" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1089.0, + 713.0, + 1089.0, + 713.0, + 1116.0, + 118.0, + 1116.0 + ], + "score": 0.98, + "text": "peak demand and plant water availability may explain" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1120.0, + 715.0, + 1120.0, + 715.0, + 1148.0, + 118.0, + 1148.0 + ], + "score": 0.98, + "text": "the high reductions relative to the remaining catchments" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 1156.0, + 715.0, + 1156.0, + 715.0, + 1182.0, + 117.0, + 1182.0 + ], + "score": 0.96, + "text": "in Group 2. In addition, the stocking density was" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1188.0, + 712.0, + 1188.0, + 712.0, + 1213.0, + 118.0, + 1213.0 + ], + "score": 0.96, + "text": "described as‘abnormally dense’by Scott et al.(2000)" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1221.0, + 715.0, + 1221.0, + 715.0, + 1246.0, + 118.0, + 1246.0 + ], + "score": 0.99, + "text": "Growthat Glendhu2wasnotablyslow(Faheyand" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1254.0, + 714.0, + 1254.0, + 714.0, + 1279.0, + 118.0, + 1279.0 + ], + "score": 0.95, + "text": "Jackson, 1997) and Lambrechtsbos A and Biesievlei are" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1289.0, + 714.0, + 1289.0, + 714.0, + 1313.0, + 119.0, + 1313.0 + ], + "score": 0.94, + "text": "described as being within sub optimalgrowth zones" + }, + { + "category_id": 15, + "poly": [ + 121.0, + 1322.0, + 714.0, + 1322.0, + 714.0, + 1346.0, + 121.0, + 1346.0 + ], + "score": 0.96, + "text": "(Scott and Smith,1997) characterised by these authors" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 1356.0, + 715.0, + 1356.0, + 715.0, + 1378.0, + 118.0, + 1378.0 + ], + "score": 1.0, + "text": "ashavingrelativelyslowresponsetimesandlesser" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 1388.0, + 578.0, + 1388.0, + 578.0, + 1412.0, + 119.0, + 1412.0 + ], + "score": 0.97, + "text": "reductions that those at more optimal sites" + }, + { + "category_id": 15, + "poly": [ + 152.0, + 590.0, + 713.0, + 590.0, + 713.0, + 616.0, + 152.0, + 616.0 + ], + "score": 0.98, + "text": "The small Australian catchments converted to pine in" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 627.0, + 715.0, + 627.0, + 715.0, + 647.0, + 119.0, + 647.0 + ], + "score": 0.99, + "text": "responsegroup1(StewartsCreek5,PineCreekand" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 657.0, + 713.0, + 657.0, + 713.0, + 683.0, + 118.0, + 683.0 + ], + "score": 0.97, + "text": "Redhill) have similar shallow soils, potential evapo-" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 691.0, + 711.0, + 691.0, + 711.0, + 714.0, + 119.0, + 714.0 + ], + "score": 0.99, + "text": "transpirationandrainfalldistribution(relativelyuni-" + }, + { + "category_id": 15, + "poly": [ + 117.0, + 724.0, + 713.0, + 724.0, + 713.0, + 749.0, + 117.0, + 749.0 + ], + "score": 0.96, + "text": "form)although Stewarts Creek is significantly wetter." + }, + { + "category_id": 15, + "poly": [ + 117.0, + 756.0, + 714.0, + 756.0, + 714.0, + 782.0, + 117.0, + 782.0 + ], + "score": 0.96, + "text": "The combination of small catchment area and the" + }, + { + "category_id": 15, + "poly": [ + 119.0, + 791.0, + 715.0, + 791.0, + 715.0, + 815.0, + 119.0, + 815.0 + ], + "score": 0.97, + "text": "increased transpirativedemand that exceedssummer" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 821.0, + 714.0, + 821.0, + 714.0, + 851.0, + 118.0, + 851.0 + ], + "score": 0.99, + "text": "and autumn rainfall and stored water results in the large" + }, + { + "category_id": 15, + "poly": [ + 118.0, + 856.0, + 635.0, + 856.0, + 635.0, + 885.0, + 118.0, + 885.0 + ], + "score": 0.99, + "text": "impact on lower flows, compared to high flows." + }, + { + "category_id": 15, + "poly": [ + 1345.0, + 196.0, + 1382.0, + 196.0, + 1382.0, + 218.0, + 1345.0, + 218.0 + ], + "score": 1.0, + "text": "263" + }, + { + "category_id": 15, + "poly": [ + 784.0, + 1354.0, + 1110.0, + 1354.0, + 1110.0, + 1379.0, + 784.0, + 1379.0 + ], + "score": 0.99, + "text": "6.Summary andconclusions" + }, + { + "category_id": 15, + "poly": [ + 469.0, + 197.0, + 1028.0, + 197.0, + 1028.0, + 220.0, + 469.0, + 220.0 + ], + "score": 0.99, + "text": "P.N.J.Laneet al./JournalofHydrology310(2005)253-265" + } + ], + "page_info": { + "page_no": 10, + "height": 2064, + "width": 1512 + } + }, + { + "layout_dets": [ + { + "category_id": 1, + "poly": [ + 129.7504425048828, + 253.22445678710938, + 731.6268920898438, + 253.22445678710938, + 731.6268920898438, + 850.9736328125, + 129.7504425048828, + 850.9736328125 + ], + "score": 0.9999896287918091 + }, + { + "category_id": 1, + "poly": [ + 130.6014862060547, + 1007.985107421875, + 731.66796875, + 1007.985107421875, + 731.66796875, + 1439.5816650390625, + 130.6014862060547, + 1439.5816650390625 + ], + "score": 0.9999825954437256 + }, + { + "category_id": 1, + "poly": [ + 130.93617248535156, + 1596.6112060546875, + 731.9217529296875, + 1596.6112060546875, + 731.9217529296875, + 1846.48779296875, + 130.93617248535156, + 1846.48779296875 + ], + "score": 0.9999703168869019 + }, + { + "category_id": 2, + "poly": [ + 128.89547729492188, + 194.6790008544922, + 167.4661407470703, + 194.6790008544922, + 167.4661407470703, + 216.0607147216797, + 128.89547729492188, + 216.0607147216797 + ], + "score": 0.9999610781669617 + }, + { + "category_id": 0, + "poly": [ + 131.96646118164062, + 943.1311645507812, + 353.8561706542969, + 943.1311645507812, + 353.8561706542969, + 973.1969604492188, + 131.96646118164062, + 973.1969604492188 + ], + "score": 0.9999556541442871 + }, + { + "category_id": 1, + "poly": [ + 785.2122192382812, + 248.98667907714844, + 1401.3350830078125, + 248.98667907714844, + 1401.3350830078125, + 1848.1571044921875, + 785.2122192382812, + 1848.1571044921875 + ], + "score": 0.9999303817749023 + }, + { + "category_id": 0, + "poly": [ + 130.318115234375, + 1534.687255859375, + 256.7127685546875, + 1534.687255859375, + 256.7127685546875, + 1561.822509765625, + 130.318115234375, + 1561.822509765625 + ], + "score": 0.9999189376831055 + }, + { + "category_id": 2, + "poly": [ + 479.6887512207031, + 194.76837158203125, + 1046.6405029296875, + 194.76837158203125, + 1046.6405029296875, + 219.2889862060547, + 479.6887512207031, + 219.2889862060547 + ], + "score": 0.9995745420455933 + }, + { + "category_id": 13, + "poly": [ + 1067, + 1022, + 1120, + 1022, + 1120, + 1049, + 1067, + 1049 + ], + "score": 0.57, + "latex": "219\\,\\mathrm{p}" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 259.0, + 729.0, + 259.0, + 729.0, + 280.0, + 133.0, + 280.0 + ], + "score": 0.98, + "text": "scales,speciesandenvironments,andhaveshownthatit" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 294.0, + 728.0, + 294.0, + 728.0, + 315.0, + 132.0, + 315.0 + ], + "score": 1.0, + "text": "providesameansofseparatingtheinfluenceofclimate" + }, + { + "category_id": 15, + "poly": [ + 134.0, + 326.0, + 727.0, + 326.0, + 727.0, + 350.0, + 134.0, + 350.0 + ], + "score": 0.96, + "text": "and vegetation on the FDCs.The modelled results" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 356.0, + 729.0, + 356.0, + 729.0, + 384.0, + 131.0, + 384.0 + ], + "score": 0.97, + "text": "showed the greatest proportional impacts were for" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 393.0, + 725.0, + 393.0, + 725.0, + 413.0, + 133.0, + 413.0 + ], + "score": 0.99, + "text": "medianandlowerflows.Theflowreductionsfromthe" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 423.0, + 729.0, + 423.0, + 729.0, + 448.0, + 132.0, + 448.0 + ], + "score": 1.0, + "text": "threesmallcatchmentsSEAustralianwerethehighest" + }, + { + "category_id": 15, + "poly": [ + 134.0, + 458.0, + 731.0, + 458.0, + 731.0, + 482.0, + 134.0, + 482.0 + ], + "score": 0.99, + "text": "and may reflect lower storages. The characterisation of" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 490.0, + 729.0, + 490.0, + 729.0, + 515.0, + 131.0, + 515.0 + ], + "score": 0.99, + "text": "thenumberofzeroflowdayswas alsosuccessfulfor" + }, + { + "category_id": 15, + "poly": [ + 130.0, + 523.0, + 729.0, + 523.0, + 729.0, + 551.0, + 130.0, + 551.0 + ], + "score": 0.98, + "text": "these catchments in indicating a significant increase in" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 556.0, + 726.0, + 556.0, + 726.0, + 581.0, + 132.0, + 581.0 + ], + "score": 0.99, + "text": "zeroflows.Theflowreductionsidentified hereprobably" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 593.0, + 728.0, + 593.0, + 728.0, + 614.0, + 132.0, + 614.0 + ], + "score": 1.0, + "text": "representamaximumeffectgiventhesizeofthe" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 625.0, + 725.0, + 625.0, + 725.0, + 648.0, + 133.0, + 648.0 + ], + "score": 1.0, + "text": "catchments,levelofafforestationandtheshallowsoils" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 657.0, + 727.0, + 657.0, + 727.0, + 681.0, + 133.0, + 681.0 + ], + "score": 0.98, + "text": "These results have yielded useful new insights on the" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 691.0, + 730.0, + 691.0, + 730.0, + 714.0, + 132.0, + 714.0 + ], + "score": 1.0, + "text": "contentiousissueofthehydrologicalimpactof" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 721.0, + 728.0, + 721.0, + 728.0, + 749.0, + 133.0, + 749.0 + ], + "score": 0.97, + "text": "afforestation. This research has led to the development" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 756.0, + 729.0, + 756.0, + 729.0, + 780.0, + 132.0, + 780.0 + ], + "score": 0.98, + "text": "of a method to assess the net impact of afforestation on" + }, + { + "category_id": 15, + "poly": [ + 130.0, + 788.0, + 727.0, + 788.0, + 727.0, + 816.0, + 130.0, + 816.0 + ], + "score": 0.98, + "text": "the flow duration curve which does not require paired-" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 825.0, + 591.0, + 825.0, + 591.0, + 849.0, + 133.0, + 849.0 + ], + "score": 0.99, + "text": "catchments to remove climatic variability." + }, + { + "category_id": 15, + "poly": [ + 165.0, + 1014.0, + 726.0, + 1014.0, + 726.0, + 1038.0, + 165.0, + 1038.0 + ], + "score": 0.94, + "text": "The authors would like tothank RoryNathan." + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1048.0, + 728.0, + 1048.0, + 728.0, + 1071.0, + 132.0, + 1071.0 + ], + "score": 0.98, + "text": "Narendra Tuteja,Tom McMahon,Geoff Podger,Rob" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 1079.0, + 725.0, + 1079.0, + 725.0, + 1105.0, + 133.0, + 1105.0 + ], + "score": 0.99, + "text": "Vertessy, Glen Walker and Peter Hairsine for particu-" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1111.0, + 728.0, + 1111.0, + 728.0, + 1139.0, + 131.0, + 1139.0 + ], + "score": 0.99, + "text": "larly helpful discussions on methodologies and reviews," + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1146.0, + 728.0, + 1146.0, + 728.0, + 1171.0, + 131.0, + 1171.0 + ], + "score": 0.96, + "text": "Richard Morton for valuable statistical advice,Dave" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1180.0, + 726.0, + 1180.0, + 726.0, + 1205.0, + 132.0, + 1205.0 + ], + "score": 0.97, + "text": "Scottfor supplying theSouthAfrican data,BarryFahey" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1213.0, + 728.0, + 1213.0, + 728.0, + 1237.0, + 132.0, + 1237.0 + ], + "score": 0.97, + "text": "for the New Zealand data, and Hancocks Victorian" + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1245.0, + 726.0, + 1245.0, + 726.0, + 1270.0, + 131.0, + 1270.0 + ], + "score": 0.97, + "text": "Plantationsfor vegetation data.Thestudywasfunded by" + }, + { + "category_id": 15, + "poly": [ + 133.0, + 1280.0, + 729.0, + 1280.0, + 729.0, + 1303.0, + 133.0, + 1303.0 + ], + "score": 1.0, + "text": "theVictorianDepartmentofNaturalResourcesand" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1314.0, + 729.0, + 1314.0, + 729.0, + 1337.0, + 132.0, + 1337.0 + ], + "score": 0.97, + "text": "EnvironmentPrivateForestryUnit,the CRC for" + }, + { + "category_id": 15, + "poly": [ + 130.0, + 1344.0, + 729.0, + 1344.0, + 729.0, + 1372.0, + 130.0, + 1372.0 + ], + "score": 0.99, + "text": "Catchment Hydrology, and the MDBC funded project" + }, + { + "category_id": 15, + "poly": [ + 135.0, + 1377.0, + 727.0, + 1377.0, + 727.0, + 1405.0, + 135.0, + 1405.0 + ], + "score": 0.98, + "text": "Integrated assessment of the effects of land use changes" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 1412.0, + 554.0, + 1412.0, + 554.0, + 1437.0, + 132.0, + 1437.0 + ], + "score": 0.96, + "text": "on water yield and salt loads'(D2013)." + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1602.0, + 729.0, + 1602.0, + 729.0, + 1625.0, + 131.0, + 1625.0 + ], + "score": 0.97, + "text": "Bosch, J.M.,1979. Treatment effects on annual and dry period" + }, + { + "category_id": 15, + "poly": [ + 164.0, + 1629.0, + 726.0, + 1629.0, + 726.0, + 1650.0, + 164.0, + 1650.0 + ], + "score": 0.98, + "text": "streamflow at CathedralPeak.SouthAfricanForestryJournal 108" + }, + { + "category_id": 15, + "poly": [ + 164.0, + 1656.0, + 224.0, + 1656.0, + 224.0, + 1677.0, + 164.0, + 1677.0 + ], + "score": 0.99, + "text": "29-37." + }, + { + "category_id": 15, + "poly": [ + 130.0, + 1684.0, + 729.0, + 1684.0, + 729.0, + 1706.0, + 130.0, + 1706.0 + ], + "score": 0.97, + "text": "Bosch, J.M., Von Gadow, K.,1990. Regulating afforestation for water" + }, + { + "category_id": 15, + "poly": [ + 164.0, + 1712.0, + 727.0, + 1712.0, + 727.0, + 1734.0, + 164.0, + 1734.0 + ], + "score": 0.97, + "text": "conservation inSouthAfrica.Suid-AfrikaanseBosboutydskrif 153," + }, + { + "category_id": 15, + "poly": [ + 165.0, + 1741.0, + 222.0, + 1741.0, + 222.0, + 1759.0, + 165.0, + 1759.0 + ], + "score": 0.98, + "text": "41-54." + }, + { + "category_id": 15, + "poly": [ + 131.0, + 1768.0, + 730.0, + 1768.0, + 730.0, + 1790.0, + 131.0, + 1790.0 + ], + "score": 0.98, + "text": "Chiew,F.H.S.,McMahon, T.A.,1993.Assessing the adequacy of" + }, + { + "category_id": 15, + "poly": [ + 165.0, + 1796.0, + 729.0, + 1796.0, + 729.0, + 1815.0, + 165.0, + 1815.0 + ], + "score": 1.0, + "text": "catchmentstreamflowyieldestimates.AustralianJournalofSoil" + }, + { + "category_id": 15, + "poly": [ + 165.0, + 1823.0, + 360.0, + 1823.0, + 360.0, + 1843.0, + 165.0, + 1843.0 + ], + "score": 0.99, + "text": "Research31,665-680." + }, + { + "category_id": 15, + "poly": [ + 129.0, + 195.0, + 167.0, + 195.0, + 167.0, + 218.0, + 129.0, + 218.0 + ], + "score": 1.0, + "text": "264" + }, + { + "category_id": 15, + "poly": [ + 132.0, + 945.0, + 352.0, + 945.0, + 352.0, + 972.0, + 132.0, + 972.0 + ], + "score": 1.0, + "text": "Acknowledgements" + }, + { + "category_id": 15, + "poly": [ + 792.0, + 253.0, + 1396.0, + 253.0, + 1396.0, + 285.0, + 792.0, + 285.0 + ], + "score": 0.98, + "text": "Cornish, P.M., Vertessy, R.A., 2001. Forest age-induced changes in" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 287.0, + 1396.0, + 287.0, + 1396.0, + 311.0, + 828.0, + 311.0 + ], + "score": 0.99, + "text": "evapotranspiration and water yield in a eucalypt forest. Journal of" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 315.0, + 1029.0, + 315.0, + 1029.0, + 340.0, + 828.0, + 340.0 + ], + "score": 0.99, + "text": "Hydrology 242,43-63." + }, + { + "category_id": 15, + "poly": [ + 794.0, + 341.0, + 1393.0, + 341.0, + 1393.0, + 368.0, + 794.0, + 368.0 + ], + "score": 0.98, + "text": "Fahey, B., Jackson, R., 1997. Hydrological impacts of converting" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 373.0, + 1394.0, + 373.0, + 1394.0, + 396.0, + 828.0, + 396.0 + ], + "score": 0.95, + "text": "native forests and grasslands to pine plantations, South" + }, + { + "category_id": 15, + "poly": [ + 825.0, + 394.0, + 1396.0, + 394.0, + 1396.0, + 428.0, + 825.0, + 428.0 + ], + "score": 0.97, + "text": "Island, New Zealand. Agricultural and Forest Meteorology 84," + }, + { + "category_id": 15, + "poly": [ + 826.0, + 428.0, + 888.0, + 428.0, + 888.0, + 453.0, + 826.0, + 453.0 + ], + "score": 1.0, + "text": "69-82." + }, + { + "category_id": 15, + "poly": [ + 795.0, + 455.0, + 1392.0, + 455.0, + 1392.0, + 480.0, + 795.0, + 480.0 + ], + "score": 0.98, + "text": "Hickel, K., 2001. The effect of pine afforestation on flow regime in" + }, + { + "category_id": 15, + "poly": [ + 823.0, + 481.0, + 1394.0, + 481.0, + 1394.0, + 513.0, + 823.0, + 513.0 + ], + "score": 0.99, + "text": " small upland catchments. Masters Thesis, University of Stuttgart," + }, + { + "category_id": 15, + "poly": [ + 822.0, + 514.0, + 889.0, + 514.0, + 889.0, + 536.0, + 822.0, + 536.0 + ], + "score": 1.0, + "text": "p.134." + }, + { + "category_id": 15, + "poly": [ + 794.0, + 542.0, + 1394.0, + 542.0, + 1394.0, + 566.0, + 794.0, + 566.0 + ], + "score": 0.98, + "text": "Holmes, J.W., Sinclair, J.A., 1986. Water yield from some afforested" + }, + { + "category_id": 15, + "poly": [ + 830.0, + 572.0, + 1394.0, + 572.0, + 1394.0, + 596.0, + 830.0, + 596.0 + ], + "score": 0.95, + "text": "catchments in Victoria. In Hydrology and Water Resources" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 600.0, + 1392.0, + 600.0, + 1392.0, + 625.0, + 828.0, + 625.0 + ], + "score": 0.99, + "text": "Symposium, Griffith University, Brisbane 25-27 November 1986," + }, + { + "category_id": 15, + "poly": [ + 824.0, + 629.0, + 937.0, + 629.0, + 937.0, + 649.0, + 824.0, + 649.0 + ], + "score": 0.96, + "text": "pp.214-218" + }, + { + "category_id": 15, + "poly": [ + 795.0, + 655.0, + 1394.0, + 655.0, + 1394.0, + 680.0, + 795.0, + 680.0 + ], + "score": 0.98, + "text": "Lane, P.N.J., Best, A.E., Hickel, K., Zhang, L., 2003. The effect" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 685.0, + 1392.0, + 685.0, + 1392.0, + 710.0, + 828.0, + 710.0 + ], + "score": 0.99, + "text": "of afforestation on flow duration curves. Cooperative Research" + }, + { + "category_id": 15, + "poly": [ + 826.0, + 712.0, + 1394.0, + 712.0, + 1394.0, + 742.0, + 826.0, + 742.0 + ], + "score": 0.96, + "text": "Centre for Catchment Hydrology Technical Report 0O3/13," + }, + { + "category_id": 15, + "poly": [ + 822.0, + 742.0, + 883.0, + 742.0, + 883.0, + 764.0, + 822.0, + 764.0 + ], + "score": 0.98, + "text": "p.25." + }, + { + "category_id": 15, + "poly": [ + 792.0, + 770.0, + 1391.0, + 770.0, + 1391.0, + 793.0, + 792.0, + 793.0 + ], + "score": 0.97, + "text": "Legates, D.R., McCabe, G.J., 1999. Evaluating the use of ^goodness-" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 798.0, + 1392.0, + 798.0, + 1392.0, + 823.0, + 828.0, + 823.0 + ], + "score": 0.97, + "text": "of-fit’ measures in hydrologic and hydroclimatic model validation." + }, + { + "category_id": 15, + "poly": [ + 828.0, + 827.0, + 1179.0, + 827.0, + 1179.0, + 850.0, + 828.0, + 850.0 + ], + "score": 0.98, + "text": "Water Resources Research 35,233-241." + }, + { + "category_id": 15, + "poly": [ + 792.0, + 851.0, + 1396.0, + 851.0, + 1396.0, + 880.0, + 792.0, + 880.0 + ], + "score": 1.0, + "text": "Lyne, V.D., Hollick, M., 1979. Stochastic time-varying rainfall-runoff" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 883.0, + 1392.0, + 883.0, + 1392.0, + 908.0, + 828.0, + 908.0 + ], + "score": 0.98, + "text": "modelling. Hydrology and Water Resources Symposium, Perth." + }, + { + "category_id": 15, + "poly": [ + 828.0, + 912.0, + 1220.0, + 912.0, + 1220.0, + 936.0, + 828.0, + 936.0 + ], + "score": 0.99, + "text": "Institution of Engineers, Australia, pp. 89-92." + }, + { + "category_id": 15, + "poly": [ + 794.0, + 940.0, + 1392.0, + 940.0, + 1392.0, + 965.0, + 794.0, + 965.0 + ], + "score": 0.98, + "text": "Nandakumar, N., Mein, R.G., 1993. Analysis of paired catchment data" + }, + { + "category_id": 15, + "poly": [ + 825.0, + 965.0, + 1396.0, + 965.0, + 1396.0, + 997.0, + 825.0, + 997.0 + ], + "score": 0.98, + "text": "to determine the hydrologic effects of changes in vegetative cover" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 998.0, + 1394.0, + 998.0, + 1394.0, + 1023.0, + 828.0, + 1023.0 + ], + "score": 0.99, + "text": "on yield. Technical Report for Project UM010, Monash University" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 1027.0, + 1066.0, + 1027.0, + 1066.0, + 1052.0, + 828.0, + 1052.0 + ], + "score": 0.97, + "text": "Dept. of Civil Engineering," + }, + { + "category_id": 15, + "poly": [ + 792.0, + 1050.0, + 1396.0, + 1050.0, + 1396.0, + 1082.0, + 792.0, + 1082.0 + ], + "score": 0.99, + "text": "Nash, J.E., Sutcliffe, J.V., 1970. River flow forecasting through" + }, + { + "category_id": 15, + "poly": [ + 826.0, + 1080.0, + 1398.0, + 1080.0, + 1398.0, + 1110.0, + 826.0, + 1110.0 + ], + "score": 0.99, + "text": "conceptual models, I, A discussion of principals. Journal of" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 1112.0, + 1041.0, + 1112.0, + 1041.0, + 1137.0, + 828.0, + 1137.0 + ], + "score": 0.98, + "text": "Hydrology 10, 282-290." + }, + { + "category_id": 15, + "poly": [ + 794.0, + 1137.0, + 1398.0, + 1137.0, + 1398.0, + 1167.0, + 794.0, + 1167.0 + ], + "score": 0.99, + "text": "Putahena, W.M., Cordery, I., 2000. Some hydrological effects of" + }, + { + "category_id": 15, + "poly": [ + 826.0, + 1167.0, + 1396.0, + 1167.0, + 1396.0, + 1197.0, + 826.0, + 1197.0 + ], + "score": 0.99, + "text": "changing forest cover from eucalyptus to Pinus radiata. Agricul-" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 1197.0, + 1188.0, + 1197.0, + 1188.0, + 1222.0, + 828.0, + 1222.0 + ], + "score": 0.99, + "text": "tural and Forest Meteorology 100, 59-72." + }, + { + "category_id": 15, + "poly": [ + 792.0, + 1223.0, + 1396.0, + 1223.0, + 1396.0, + 1253.0, + 792.0, + 1253.0 + ], + "score": 0.98, + "text": "Roberts, S., Vertessy, R.A., Grayson, R.G., 2001. Transpiration from" + }, + { + "category_id": 15, + "poly": [ + 825.0, + 1250.0, + 1396.0, + 1250.0, + 1396.0, + 1282.0, + 825.0, + 1282.0 + ], + "score": 0.99, + "text": "Eucalyptus sieberi (L. Johnson) forests of different age. Forest" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 1280.0, + 1184.0, + 1280.0, + 1184.0, + 1310.0, + 828.0, + 1310.0 + ], + "score": 0.98, + "text": "Ecology and Management 143, 153-161." + }, + { + "category_id": 15, + "poly": [ + 792.0, + 1307.0, + 1396.0, + 1307.0, + 1396.0, + 1338.0, + 792.0, + 1338.0 + ], + "score": 0.99, + "text": "Scott, D.F., Smith, R.E., 1997. Preliminary empirical models to predict" + }, + { + "category_id": 15, + "poly": [ + 825.0, + 1335.0, + 1394.0, + 1335.0, + 1394.0, + 1367.0, + 825.0, + 1367.0 + ], + "score": 0.99, + "text": "reductions in total and low flows resulting from afforestation." + }, + { + "category_id": 15, + "poly": [ + 826.0, + 1365.0, + 1045.0, + 1365.0, + 1045.0, + 1392.0, + 826.0, + 1392.0 + ], + "score": 0.96, + "text": "Water S.A. 23, 135-140." + }, + { + "category_id": 15, + "poly": [ + 795.0, + 1393.0, + 1391.0, + 1393.0, + 1391.0, + 1418.0, + 795.0, + 1418.0 + ], + "score": 0.93, + "text": "Scott, D.F., Prinsloo, F.W., Moses, G., Mehlomakulu, M.," + }, + { + "category_id": 15, + "poly": [ + 826.0, + 1423.0, + 1392.0, + 1423.0, + 1392.0, + 1448.0, + 826.0, + 1448.0 + ], + "score": 0.95, + "text": "Simmers, A.D.A., 2000. Area-analysis of the South African" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 1452.0, + 1392.0, + 1452.0, + 1392.0, + 1477.0, + 828.0, + 1477.0 + ], + "score": 0.96, + "text": "catchment afforestation experimental data. WRC Report" + }, + { + "category_id": 15, + "poly": [ + 827.0, + 1478.0, + 954.0, + 1478.0, + 954.0, + 1507.0, + 827.0, + 1507.0 + ], + "score": 0.99, + "text": "No. 810/1/00." + }, + { + "category_id": 15, + "poly": [ + 792.0, + 1505.0, + 1396.0, + 1505.0, + 1396.0, + 1537.0, + 792.0, + 1537.0 + ], + "score": 0.99, + "text": "Sikka, A.K., Samra, J.S., Sharda, V.N., Samraj, P., Lakshmanan, V.," + }, + { + "category_id": 15, + "poly": [ + 826.0, + 1537.0, + 1394.0, + 1537.0, + 1394.0, + 1563.0, + 826.0, + 1563.0 + ], + "score": 0.99, + "text": "2003. Low fow and high responses to converting natural grassland" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 1567.0, + 1396.0, + 1567.0, + 1396.0, + 1592.0, + 828.0, + 1592.0 + ], + "score": 0.98, + "text": "into bluegum (Eucalyptus globulus) in Ningiris watersheds of" + }, + { + "category_id": 15, + "poly": [ + 825.0, + 1590.0, + 1234.0, + 1590.0, + 1234.0, + 1622.0, + 825.0, + 1622.0 + ], + "score": 0.99, + "text": "South India. Journal of Hydrology 270, 12-26." + }, + { + "category_id": 15, + "poly": [ + 795.0, + 1624.0, + 1392.0, + 1624.0, + 1392.0, + 1648.0, + 795.0, + 1648.0 + ], + "score": 0.99, + "text": "Smakhtin, V.U., 1999. A concept of pragmatic hydrological time series" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 1652.0, + 1394.0, + 1652.0, + 1394.0, + 1677.0, + 828.0, + 1677.0 + ], + "score": 0.99, + "text": "modelling and its application in South African context. In Ninth" + }, + { + "category_id": 15, + "poly": [ + 825.0, + 1677.0, + 1396.0, + 1677.0, + 1396.0, + 1709.0, + 825.0, + 1709.0 + ], + "score": 0.97, + "text": "South African National Hydrology Symposium, 29-30 November" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 1707.0, + 966.0, + 1707.0, + 966.0, + 1737.0, + 828.0, + 1737.0 + ], + "score": 1.0, + "text": "1999, pp. 1-11." + }, + { + "category_id": 15, + "poly": [ + 792.0, + 1733.0, + 1398.0, + 1733.0, + 1398.0, + 1765.0, + 792.0, + 1765.0 + ], + "score": 0.99, + "text": "Smakhtin, V.U., 2001. Low flow hydrology: a review. Journal of" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 1765.0, + 1050.0, + 1765.0, + 1050.0, + 1790.0, + 828.0, + 1790.0 + ], + "score": 0.98, + "text": "Hydrology 240, 147-186." + }, + { + "category_id": 15, + "poly": [ + 794.0, + 1790.0, + 1396.0, + 1790.0, + 1396.0, + 1822.0, + 794.0, + 1822.0 + ], + "score": 0.98, + "text": "Van Lill, W.S., Kruger, F.J., Van Wyk, D.B., 1980. The effect of" + }, + { + "category_id": 15, + "poly": [ + 828.0, + 1824.0, + 1394.0, + 1824.0, + 1394.0, + 1848.0, + 828.0, + 1848.0 + ], + "score": 0.98, + "text": "afforestation with Eucalyptus grandis Hill ex Maiden and Pinus" + }, + { + "category_id": 15, + "poly": [ + 130.0, + 1533.0, + 257.0, + 1533.0, + 257.0, + 1561.0, + 130.0, + 1561.0 + ], + "score": 1.0, + "text": "References" + }, + { + "category_id": 15, + "poly": [ + 482.0, + 196.0, + 1041.0, + 196.0, + 1041.0, + 220.0, + 482.0, + 220.0 + ], + "score": 0.97, + "text": "P.N.J.Lane et al./ Journal ofHydrology 310(2005)253-265" + } + ], + "page_info": { + "page_no": 11, + "height": 2064, + "width": 1512 + } + }, + { + "layout_dets": [ + { + "category_id": 1, + "poly": [ + 775.5120849609375, + 251.32174682617188, + 1385.47265625, + 251.32174682617188, + 1385.47265625, + 615.240478515625, + 775.5120849609375, + 615.240478515625 + ], + "score": 0.9999982714653015 + }, + { + "category_id": 1, + "poly": [ + 116.47637939453125, + 255.4496612548828, + 719.7019653320312, + 255.4496612548828, + 719.7019653320312, + 616.0682373046875, + 116.47637939453125, + 616.0682373046875 + ], + "score": 0.9999979734420776 + }, + { + "category_id": 2, + "poly": [ + 1346.1534423828125, + 194.16160583496094, + 1382.1688232421875, + 194.16160583496094, + 1382.1688232421875, + 217.01524353027344, + 1346.1534423828125, + 217.01524353027344 + ], + "score": 0.9999814033508301 + }, + { + "category_id": 2, + "poly": [ + 465.8272705078125, + 194.28224182128906, + 1032.5506591796875, + 194.28224182128906, + 1032.5506591796875, + 220.2495880126953, + 465.8272705078125, + 220.2495880126953 + ], + "score": 0.9999077320098877 + }, + { + "category_id": 15, + "poly": [ + 785.0, + 257.0, + 1379.0, + 257.0, + 1379.0, + 280.0, + 785.0, + 280.0 + ], + "score": 0.97, + "text": "Vogel, R.M.,Fennessey, N.M., 1994. Flow duration curves. 1. New" + }, + { + "category_id": 15, + "poly": [ + 815.0, + 284.0, + 1381.0, + 284.0, + 1381.0, + 309.0, + 815.0, + 309.0 + ], + "score": 0.97, + "text": "interpretation and confidence intervals. Journal of Water Planning" + }, + { + "category_id": 15, + "poly": [ + 816.0, + 312.0, + 1123.0, + 312.0, + 1123.0, + 335.0, + 816.0, + 335.0 + ], + "score": 0.96, + "text": "and Management 120 (4),485-504." + }, + { + "category_id": 15, + "poly": [ + 782.0, + 339.0, + 1383.0, + 339.0, + 1383.0, + 366.0, + 782.0, + 366.0 + ], + "score": 0.97, + "text": "Whitehead, D., Beadle C.L., 2004. Physiological regulation of" + }, + { + "category_id": 15, + "poly": [ + 815.0, + 368.0, + 1379.0, + 368.0, + 1379.0, + 394.0, + 815.0, + 394.0 + ], + "score": 1.0, + "text": "productivity and water use in Eucalyptus: a review. Forest Ecology" + }, + { + "category_id": 15, + "poly": [ + 815.0, + 396.0, + 1098.0, + 396.0, + 1098.0, + 420.0, + 815.0, + 420.0 + ], + "score": 0.99, + "text": "and Management, 193, 113-140." + }, + { + "category_id": 15, + "poly": [ + 785.0, + 425.0, + 1382.0, + 425.0, + 1382.0, + 445.0, + 785.0, + 445.0 + ], + "score": 0.97, + "text": "Zhang,L.,Dawes,W.R.,Walker,G.R.,1999.Predicting the effect of" + }, + { + "category_id": 15, + "poly": [ + 817.0, + 454.0, + 1378.0, + 454.0, + 1378.0, + 474.0, + 817.0, + 474.0 + ], + "score": 1.0, + "text": "vegetationchangesoncatchmentaveragewaterbalance.Coop-" + }, + { + "category_id": 15, + "poly": [ + 814.0, + 479.0, + 1380.0, + 479.0, + 1380.0, + 501.0, + 814.0, + 501.0 + ], + "score": 0.97, + "text": "erative ResearchCentre for Catchment HydrologyTechnical" + }, + { + "category_id": 15, + "poly": [ + 816.0, + 508.0, + 988.0, + 508.0, + 988.0, + 529.0, + 816.0, + 529.0 + ], + "score": 0.99, + "text": "Report99/12,p.35." + }, + { + "category_id": 15, + "poly": [ + 783.0, + 533.0, + 1381.0, + 533.0, + 1381.0, + 558.0, + 783.0, + 558.0 + ], + "score": 0.97, + "text": "Zhang, L., Dawes, W.R.,Walker, G.R.,2001.Response of mean" + }, + { + "category_id": 15, + "poly": [ + 815.0, + 562.0, + 1382.0, + 562.0, + 1382.0, + 586.0, + 815.0, + 586.0 + ], + "score": 0.98, + "text": "annual evapotranspiration to vegetation changes at catchment" + }, + { + "category_id": 15, + "poly": [ + 815.0, + 590.0, + 1217.0, + 590.0, + 1217.0, + 610.0, + 815.0, + 610.0 + ], + "score": 0.99, + "text": "scale.WaterResourcesResearch37,701-708." + }, + { + "category_id": 15, + "poly": [ + 149.0, + 257.0, + 715.0, + 257.0, + 715.0, + 280.0, + 149.0, + 280.0 + ], + "score": 0.97, + "text": "patula Schlect.et Cham.on streamflow fromexperimental" + }, + { + "category_id": 15, + "poly": [ + 151.0, + 284.0, + 712.0, + 284.0, + 712.0, + 307.0, + 151.0, + 307.0 + ], + "score": 0.98, + "text": "catchments at Mokubulaan, Transvaal. Journal of Hydrology 48" + }, + { + "category_id": 15, + "poly": [ + 153.0, + 314.0, + 230.0, + 314.0, + 230.0, + 333.0, + 153.0, + 333.0 + ], + "score": 1.0, + "text": "107-118." + }, + { + "category_id": 15, + "poly": [ + 117.0, + 339.0, + 717.0, + 339.0, + 717.0, + 363.0, + 117.0, + 363.0 + ], + "score": 0.98, + "text": "Van Wyk, D.B.,1987. Some effects of afforestation on streamflow" + }, + { + "category_id": 15, + "poly": [ + 149.0, + 368.0, + 715.0, + 368.0, + 715.0, + 391.0, + 149.0, + 391.0 + ], + "score": 0.97, + "text": "in the Western Cape Province, South Africa.Water S.A. 13," + }, + { + "category_id": 15, + "poly": [ + 149.0, + 395.0, + 210.0, + 395.0, + 210.0, + 417.0, + 149.0, + 417.0 + ], + "score": 1.0, + "text": "31-36." + }, + { + "category_id": 15, + "poly": [ + 117.0, + 422.0, + 715.0, + 422.0, + 715.0, + 449.0, + 117.0, + 449.0 + ], + "score": 0.96, + "text": "Vertessy, R.A., Bessard, Y., 1999. Anticipating the negative" + }, + { + "category_id": 15, + "poly": [ + 151.0, + 452.0, + 717.0, + 452.0, + 717.0, + 476.0, + 151.0, + 476.0 + ], + "score": 0.97, + "text": "hydrologic effects of plantation expansion: results from a" + }, + { + "category_id": 15, + "poly": [ + 149.0, + 477.0, + 716.0, + 477.0, + 716.0, + 503.0, + 149.0, + 503.0 + ], + "score": 1.0, + "text": "GIS-based analysis on the Murrumbidgee Basin, in: Croke, J.," + }, + { + "category_id": 15, + "poly": [ + 149.0, + 507.0, + 717.0, + 507.0, + 717.0, + 530.0, + 149.0, + 530.0 + ], + "score": 0.99, + "text": "Lane, P.N.J. (Eds.), Forest Management for Water Quality and" + }, + { + "category_id": 15, + "poly": [ + 152.0, + 534.0, + 716.0, + 534.0, + 716.0, + 558.0, + 152.0, + 558.0 + ], + "score": 0.99, + "text": "Quantity: Proceedings of the 2nd Erosion in Forests Meeting" + }, + { + "category_id": 15, + "poly": [ + 151.0, + 563.0, + 716.0, + 563.0, + 716.0, + 584.0, + 151.0, + 584.0 + ], + "score": 0.98, + "text": "CooperativeResearchCentre for Catchment Hydrology,Report" + }, + { + "category_id": 15, + "poly": [ + 149.0, + 588.0, + 299.0, + 588.0, + 299.0, + 615.0, + 149.0, + 615.0 + ], + "score": 0.97, + "text": "99/6,Pp.69-73." + }, + { + "category_id": 15, + "poly": [ + 1345.0, + 196.0, + 1382.0, + 196.0, + 1382.0, + 219.0, + 1345.0, + 219.0 + ], + "score": 1.0, + "text": "265" + }, + { + "category_id": 15, + "poly": [ + 470.0, + 197.0, + 1026.0, + 197.0, + 1026.0, + 220.0, + 470.0, + 220.0 + ], + "score": 0.98, + "text": "P.N.J.Laneet al./JournalofHydrology310(2005)253-265" + } + ], + "page_info": { + "page_no": 12, + "height": 2064, + "width": 1512 + } + } +] diff --git a/tests/test_model/assets/test_02.pdf b/tests/test_model/assets/test_02.pdf new file mode 100644 index 00000000..c9405d62 Binary files /dev/null and b/tests/test_model/assets/test_02.pdf differ diff --git a/tests/test_model/test_magic_model.py b/tests/test_model/test_magic_model.py new file mode 100644 index 00000000..4fc4139c --- /dev/null +++ b/tests/test_model/test_magic_model.py @@ -0,0 +1,31 @@ +import json + +from magic_pdf.data.read_api import read_local_pdfs +from magic_pdf.model.magic_model import MagicModel + + +def test_magic_model_image_v2(): + datasets = read_local_pdfs('tests/test_model/assets/test_01.pdf') + with open('tests/test_model/assets/test_01.model.json') as f: + model_json = json.load(f) + + magic_model = MagicModel(model_json, datasets[0]) + + imgs = magic_model.get_imgs_v2(0) + print(imgs) + + tables = magic_model.get_tables_v2(0) + print(tables) + + +def test_magic_model_table_v2(): + datasets = read_local_pdfs('tests/test_model/assets/test_02.pdf') + with open('tests/test_model/assets/test_02.model.json') as f: + model_json = json.load(f) + + magic_model = MagicModel(model_json, datasets[0]) + tables = magic_model.get_tables_v2(5) + print(tables) + + tables = magic_model.get_tables_v2(8) + print(tables)