> rec_box;
+ // html tag
+ int step_start_idx = (batch_idx * structure_probs_shape[1] + step_idx) *
+ structure_probs_shape[2];
+ char_idx = int(Utility::argmax(
+ &structure_probs[step_start_idx],
+ &structure_probs[step_start_idx + structure_probs_shape[2]]));
+ char_score = float(*std::max_element(
+ &structure_probs[step_start_idx],
+ &structure_probs[step_start_idx + structure_probs_shape[2]]));
+ html_tag = this->label_list_[char_idx];
+
+ if (step_idx > 0 && html_tag == this->end) {
+ break;
+ }
+ if (html_tag == this->beg) {
+ continue;
+ }
+ count += 1;
+ score += char_score;
+ rec_html_tags.push_back(html_tag);
+ // box
+ if (html_tag == "| " || html_tag == " | point(2, 0);
+ step_start_idx = (batch_idx * structure_probs_shape[1] + step_idx) *
+ loc_preds_shape[2] +
+ point_idx;
+ point[0] = int(loc_preds[step_start_idx] * width_list[batch_idx]);
+ point[1] =
+ int(loc_preds[step_start_idx + 1] * height_list[batch_idx]);
+ rec_box.push_back(point);
+ }
+ rec_boxes.push_back(rec_box);
+ }
+ }
+ score /= count;
+ if (isnan(score) || rec_boxes.size() == 0 || rec_html_tags.size() == 0) {
+ score = -1;
+ }
+ rec_scores.push_back(score);
+ rec_boxes_batch.push_back(rec_boxes);
+ rec_html_tag_batch.push_back(rec_html_tags);
+ }
+}
+
} // namespace PaddleOCR
diff --git a/deploy/cpp_infer/src/preprocess_op.cpp b/deploy/cpp_infer/src/preprocess_op.cpp
index fff49ba2c2..ac185e22d6 100644
--- a/deploy/cpp_infer/src/preprocess_op.cpp
+++ b/deploy/cpp_infer/src/preprocess_op.cpp
@@ -69,18 +69,28 @@ void Normalize::Run(cv::Mat *im, const std::vector &mean,
}
void ResizeImgType0::Run(const cv::Mat &img, cv::Mat &resize_img,
- int max_size_len, float &ratio_h, float &ratio_w,
- bool use_tensorrt) {
+ string limit_type, int limit_side_len, float &ratio_h,
+ float &ratio_w, bool use_tensorrt) {
int w = img.cols;
int h = img.rows;
-
float ratio = 1.f;
- int max_wh = w >= h ? w : h;
- if (max_wh > max_size_len) {
- if (h > w) {
- ratio = float(max_size_len) / float(h);
- } else {
- ratio = float(max_size_len) / float(w);
+ if (limit_type == "min") {
+ int min_wh = min(h, w);
+ if (min_wh < limit_side_len) {
+ if (h < w) {
+ ratio = float(limit_side_len) / float(h);
+ } else {
+ ratio = float(limit_side_len) / float(w);
+ }
+ }
+ } else {
+ int max_wh = max(h, w);
+ if (max_wh > limit_side_len) {
+ if (h > w) {
+ ratio = float(limit_side_len) / float(h);
+ } else {
+ ratio = float(limit_side_len) / float(w);
+ }
}
}
@@ -143,4 +153,26 @@ void ClsResizeImg::Run(const cv::Mat &img, cv::Mat &resize_img,
}
}
+void TableResizeImg::Run(const cv::Mat &img, cv::Mat &resize_img,
+ const int max_len) {
+ int w = img.cols;
+ int h = img.rows;
+
+ int max_wh = w >= h ? w : h;
+ float ratio = w >= h ? float(max_len) / float(w) : float(max_len) / float(h);
+
+ int resize_h = int(float(h) * ratio);
+ int resize_w = int(float(w) * ratio);
+
+ cv::resize(img, resize_img, cv::Size(resize_w, resize_h));
+}
+
+void TablePadImg::Run(const cv::Mat &img, cv::Mat &resize_img,
+ const int max_len) {
+ int w = img.cols;
+ int h = img.rows;
+ cv::copyMakeBorder(img, resize_img, 0, max_len - h, 0, max_len - w,
+ cv::BORDER_CONSTANT, cv::Scalar(0, 0, 0));
+}
+
} // namespace PaddleOCR
diff --git a/deploy/cpp_infer/src/structure_table.cpp b/deploy/cpp_infer/src/structure_table.cpp
new file mode 100644
index 0000000000..bbc32580e4
--- /dev/null
+++ b/deploy/cpp_infer/src/structure_table.cpp
@@ -0,0 +1,158 @@
+// Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved.
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+// http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+#include
+
+namespace PaddleOCR {
+
+void StructureTableRecognizer::Run(
+ std::vector img_list,
+ std::vector> &structure_html_tags,
+ std::vector &structure_scores,
+ std::vector>>> &structure_boxes,
+ std::vector ×) {
+ std::chrono::duration preprocess_diff =
+ std::chrono::steady_clock::now() - std::chrono::steady_clock::now();
+ std::chrono::duration inference_diff =
+ std::chrono::steady_clock::now() - std::chrono::steady_clock::now();
+ std::chrono::duration postprocess_diff =
+ std::chrono::steady_clock::now() - std::chrono::steady_clock::now();
+
+ int img_num = img_list.size();
+ for (int beg_img_no = 0; beg_img_no < img_num;
+ beg_img_no += this->table_batch_num_) {
+ // preprocess
+ auto preprocess_start = std::chrono::steady_clock::now();
+ int end_img_no = min(img_num, beg_img_no + this->table_batch_num_);
+ int batch_num = end_img_no - beg_img_no;
+ std::vector norm_img_batch;
+ std::vector width_list;
+ std::vector height_list;
+ for (int ino = beg_img_no; ino < end_img_no; ino++) {
+ cv::Mat srcimg;
+ img_list[ino].copyTo(srcimg);
+ cv::Mat resize_img;
+ cv::Mat pad_img;
+ this->resize_op_.Run(srcimg, resize_img, this->table_max_len_);
+ this->normalize_op_.Run(&resize_img, this->mean_, this->scale_,
+ this->is_scale_);
+ this->pad_op_.Run(resize_img, pad_img, this->table_max_len_);
+ norm_img_batch.push_back(pad_img);
+ width_list.push_back(srcimg.cols);
+ height_list.push_back(srcimg.rows);
+ }
+
+ std::vector input(
+ batch_num * 3 * this->table_max_len_ * this->table_max_len_, 0.0f);
+ this->permute_op_.Run(norm_img_batch, input.data());
+ auto preprocess_end = std::chrono::steady_clock::now();
+ preprocess_diff += preprocess_end - preprocess_start;
+ // inference.
+ auto input_names = this->predictor_->GetInputNames();
+ auto input_t = this->predictor_->GetInputHandle(input_names[0]);
+ input_t->Reshape(
+ {batch_num, 3, this->table_max_len_, this->table_max_len_});
+ auto inference_start = std::chrono::steady_clock::now();
+ input_t->CopyFromCpu(input.data());
+ this->predictor_->Run();
+ auto output_names = this->predictor_->GetOutputNames();
+ auto output_tensor0 = this->predictor_->GetOutputHandle(output_names[0]);
+ auto output_tensor1 = this->predictor_->GetOutputHandle(output_names[1]);
+ std::vector predict_shape0 = output_tensor0->shape();
+ std::vector predict_shape1 = output_tensor1->shape();
+
+ int out_num0 = std::accumulate(predict_shape0.begin(), predict_shape0.end(),
+ 1, std::multiplies());
+ int out_num1 = std::accumulate(predict_shape1.begin(), predict_shape1.end(),
+ 1, std::multiplies());
+ std::vector loc_preds;
+ std::vector structure_probs;
+ loc_preds.resize(out_num0);
+ structure_probs.resize(out_num1);
+
+ output_tensor0->CopyToCpu(loc_preds.data());
+ output_tensor1->CopyToCpu(structure_probs.data());
+ auto inference_end = std::chrono::steady_clock::now();
+ inference_diff += inference_end - inference_start;
+ // postprocess
+ auto postprocess_start = std::chrono::steady_clock::now();
+ std::vector> structure_html_tag_batch;
+ std::vector structure_score_batch;
+ std::vector>>>
+ structure_boxes_batch;
+ this->post_processor_.Run(loc_preds, structure_probs, structure_score_batch,
+ predict_shape0, predict_shape1,
+ structure_html_tag_batch, structure_boxes_batch,
+ width_list, height_list);
+ for (int m = 0; m < predict_shape0[0]; m++) {
+
+ structure_html_tag_batch[m].insert(structure_html_tag_batch[m].begin(),
+ "");
+ structure_html_tag_batch[m].insert(structure_html_tag_batch[m].begin(),
+ "");
+ structure_html_tag_batch[m].insert(structure_html_tag_batch[m].begin(),
+ "");
+ structure_html_tag_batch[m].push_back(" ");
+ structure_html_tag_batch[m].push_back("");
+ structure_html_tag_batch[m].push_back("");
+ structure_html_tags.push_back(structure_html_tag_batch[m]);
+ structure_scores.push_back(structure_score_batch[m]);
+ structure_boxes.push_back(structure_boxes_batch[m]);
+ }
+ auto postprocess_end = std::chrono::steady_clock::now();
+ postprocess_diff += postprocess_end - postprocess_start;
+ times.push_back(double(preprocess_diff.count() * 1000));
+ times.push_back(double(inference_diff.count() * 1000));
+ times.push_back(double(postprocess_diff.count() * 1000));
+ }
+}
+
+void StructureTableRecognizer::LoadModel(const std::string &model_dir) {
+ AnalysisConfig config;
+ config.SetModel(model_dir + "/inference.pdmodel",
+ model_dir + "/inference.pdiparams");
+
+ if (this->use_gpu_) {
+ config.EnableUseGpu(this->gpu_mem_, this->gpu_id_);
+ if (this->use_tensorrt_) {
+ auto precision = paddle_infer::Config::Precision::kFloat32;
+ if (this->precision_ == "fp16") {
+ precision = paddle_infer::Config::Precision::kHalf;
+ }
+ if (this->precision_ == "int8") {
+ precision = paddle_infer::Config::Precision::kInt8;
+ }
+ config.EnableTensorRtEngine(1 << 20, 10, 3, precision, false, false);
+ }
+ } else {
+ config.DisableGpu();
+ if (this->use_mkldnn_) {
+ config.EnableMKLDNN();
+ }
+ config.SetCpuMathLibraryNumThreads(this->cpu_math_library_num_threads_);
+ }
+
+ // false for zero copy tensor
+ config.SwitchUseFeedFetchOps(false);
+ // true for multiple input
+ config.SwitchSpecifyInputNames(true);
+
+ config.SwitchIrOptim(true);
+
+ config.EnableMemoryOptim();
+ config.DisableGlogInfo();
+
+ this->predictor_ = CreatePredictor(config);
+}
+} // namespace PaddleOCR
diff --git a/deploy/cpp_infer/src/utility.cpp b/deploy/cpp_infer/src/utility.cpp
index 45b8104626..4bfc1d091d 100644
--- a/deploy/cpp_infer/src/utility.cpp
+++ b/deploy/cpp_infer/src/utility.cpp
@@ -248,4 +248,33 @@ void Utility::print_result(const std::vector &ocr_result) {
std::cout << std::endl;
}
}
+
+cv::Mat Utility::crop_image(cv::Mat &img, std::vector &area) {
+ cv::Mat crop_im;
+ int crop_x1 = std::max(0, area[0]);
+ int crop_y1 = std::max(0, area[1]);
+ int crop_x2 = std::min(img.cols - 1, area[2] - 1);
+ int crop_y2 = std::min(img.rows - 1, area[3] - 1);
+
+ crop_im = cv::Mat::zeros(area[3] - area[1], area[2] - area[0], 16);
+ cv::Mat crop_im_window =
+ crop_im(cv::Range(crop_y1 - area[1], crop_y2 + 1 - area[1]),
+ cv::Range(crop_x1 - area[0], crop_x2 + 1 - area[0]));
+ cv::Mat roi_img =
+ img(cv::Range(crop_y1, crop_y2 + 1), cv::Range(crop_x1, crop_x2 + 1));
+ crop_im_window += roi_img;
+ return crop_im;
+}
+
+void Utility::sorted_boxes(std::vector &ocr_result) {
+ std::sort(ocr_result.begin(), ocr_result.end(), Utility::comparison_box);
+
+ for (int i = 0; i < ocr_result.size() - 1; i++) {
+ if (abs(ocr_result[i + 1].box[0][1] - ocr_result[i].box[0][1]) < 10 &&
+ (ocr_result[i + 1].box[0][0] < ocr_result[i].box[0][0])) {
+ std::swap(ocr_result[i], ocr_result[i + 1]);
+ }
+ }
+}
+
} // namespace PaddleOCR
\ No newline at end of file
diff --git a/ppstructure/table/matcher.py b/ppstructure/table/matcher.py
index 6884ea3c48..9c5bd2630f 100755
--- a/ppstructure/table/matcher.py
+++ b/ppstructure/table/matcher.py
@@ -62,8 +62,8 @@ class TableMatch:
def __call__(self, structure_res, dt_boxes, rec_res):
pred_structures, pred_bboxes = structure_res
if self.filter_ocr_result:
- dt_boxes, rec_res = self.filter_ocr_result(pred_bboxes, dt_boxes,
- rec_res)
+ dt_boxes, rec_res = self._filter_ocr_result(pred_bboxes, dt_boxes,
+ rec_res)
matched_index = self.match_result(dt_boxes, pred_bboxes)
if self.use_master:
pred_html, pred = self.get_pred_html_master(pred_structures,
@@ -179,7 +179,7 @@ class TableMatch:
html = deal_bb(html)
return html, end_html
- def filter_ocr_result(self, pred_bboxes, dt_boxes, rec_res):
+ def _filter_ocr_result(self, pred_bboxes, dt_boxes, rec_res):
y1 = pred_bboxes[:, 1::2].min()
new_dt_boxes = []
new_rec_res = []
diff --git a/ppstructure/table/predict_table.py b/ppstructure/table/predict_table.py
index 57f0fec017..b0c7ef589f 100644
--- a/ppstructure/table/predict_table.py
+++ b/ppstructure/table/predict_table.py
@@ -70,7 +70,7 @@ class TableSystem(object):
if args.table_algorithm in ['TableMaster']:
self.match = TableMasterMatcher()
else:
- self.match = TableMatch()
+ self.match = TableMatch(filter_ocr_result=True)
self.benchmark = args.benchmark
self.predictor, self.input_tensor, self.output_tensors, self.config = utility.create_predictor(
From 24d917d2108706e4b1e1e9952ad81d56defe1b0d Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Wed, 10 Aug 2022 06:27:21 +0000
Subject: [PATCH 09/49] update cpp infer doc
---
deploy/cpp_infer/readme.md | 40 +++++++++++++++++++--
deploy/cpp_infer/readme_ch.md | 46 +++++++++++++++++++++---
deploy/cpp_infer/src/main.cpp | 4 +--
deploy/cpp_infer/src/paddlestructure.cpp | 4 +--
4 files changed, 82 insertions(+), 12 deletions(-)
diff --git a/deploy/cpp_infer/readme.md b/deploy/cpp_infer/readme.md
index a87db7e659..545924c5ce 100644
--- a/deploy/cpp_infer/readme.md
+++ b/deploy/cpp_infer/readme.md
@@ -171,6 +171,9 @@ inference/
|-- cls
| |--inference.pdiparams
| |--inference.pdmodel
+|-- table
+| |--inference.pdiparams
+| |--inference.pdmodel
```
@@ -275,6 +278,22 @@ Specifically,
--cls=true \
```
+
+##### 7. table
+```shell
+./build/ppocr --det_model_dir=inference/det_db \
+ --rec_model_dir=inference/rec_rcnn \
+ --cls_model_dir=inference/cls \
+ --table_model_dir=inference/table \
+ --image_dir=../../ppstructure/docs/table/table.jpg \
+ --use_angle_cls=true \
+ --det=true \
+ --rec=true \
+ --cls=true \
+ --type=structure \
+ --table=true
+```
+
More parameters are as follows,
- Common parameters
@@ -293,9 +312,9 @@ More parameters are as follows,
|parameter|data type|default|meaning|
| :---: | :---: | :---: | :---: |
-|det|bool|true|前向是否执行文字检测|
-|rec|bool|true|前向是否执行文字识别|
-|cls|bool|false|前向是否执行文字方向分类|
+|det|bool|true|Whether to perform text detection in the forward direction|
+|rec|bool|true|Whether to perform text recognition in the forward direction|
+|cls|bool|false|Whether to perform text direction classification in the forward direction|
- Detection related parameters
@@ -329,6 +348,15 @@ More parameters are as follows,
|rec_img_h|int|48|image height of recognition|
|rec_img_w|int|320|image width of recognition|
+- Table recognition related parameters
+
+|parameter|data type|default|meaning|
+| :---: | :---: | :---: | :---: |
+|table_model_dir|string|-|Address of table recognition inference model|
+|table_char_dict_path|string|../../ppocr/utils/dict/table_structure_dict.txt|dictionary file|
+|table_max_len|int|488|The size of the long side of the input image of the table recognition model, the final input image size of the network is(table_max_len,table_max_len)|
+
+
* Multi-language inference is also supported in PaddleOCR, you can refer to [recognition tutorial](../../doc/doc_en/recognition_en.md) for more supported languages and models in PaddleOCR. Specifically, if you want to infer using multi-language models, you just need to modify values of `rec_char_dict_path` and `rec_model_dir`.
@@ -344,6 +372,12 @@ predict img: ../../doc/imgs/12.jpg
The detection visualized image saved in ./output//12.jpg
```
+- table
+
+```bash
+predict img: ../../ppstructure/docs/table/table.jpg
+0 type: table, region: [0,0,371,293], res: | Methods | R | P | F | FPS | | SegLink [26] | 70.0 | 86.0 | 77.0 | 8.9 | | PixelLink [4] | 73.2 | 83.0 | 77.8 | - | | TextSnake [18] | 73.9 | 83.2 | 78.3 | 1.1 | | TextField [37] | 75.9 | 87.4 | 81.3 | 5.2 | | MSR[38] | 76.7 | 87.4 | 81.7 | - | | FTSN [3] | 77.1 | 87.6 | 82.0 | - | | LSE[30] | 81.7 | 84.2 | 82.9 | - | | CRAFT [2] | 78.2 | 88.2 | 82.9 | 8.6 | | MCN [16] | 79 | 88 | 83 | - | | ATRR[35] | 82.1 | 85.2 | 83.6 | - | | PAN [34] | 83.8 | 84.4 | 84.1 | 30.2 | | DB[12] | 79.2 | 91.5 | 84.9 | 32.0 | | DRRG [41] | 82.30 | 88.05 | 85.08 | - | | Ours (SynText) | 80.68 | 85.40 | 82.97 | 12.68 | | Ours (MLT-17) | 84.54 | 86.62 | 85.57 | 12.31 |
+```
## 3. FAQ
diff --git a/deploy/cpp_infer/readme_ch.md b/deploy/cpp_infer/readme_ch.md
index 8c334851c0..fb994a5b41 100644
--- a/deploy/cpp_infer/readme_ch.md
+++ b/deploy/cpp_infer/readme_ch.md
@@ -181,6 +181,9 @@ inference/
|-- cls
| |--inference.pdiparams
| |--inference.pdmodel
+|-- table
+| |--inference.pdiparams
+| |--inference.pdmodel
```
@@ -285,6 +288,21 @@ CUDNN_LIB_DIR=/your_cudnn_lib_dir
--cls=true \
```
+##### 7. 表格识别
+```shell
+./build/ppocr --det_model_dir=inference/det_db \
+ --rec_model_dir=inference/rec_rcnn \
+ --cls_model_dir=inference/cls \
+ --table_model_dir=inference/table \
+ --image_dir=../../ppstructure/docs/table/table.jpg \
+ --use_angle_cls=true \
+ --det=true \
+ --rec=true \
+ --cls=true \
+ --type=structure \
+ --table=true
+```
+
更多支持的可调节参数解释如下:
- 通用参数
@@ -328,21 +346,32 @@ CUDNN_LIB_DIR=/your_cudnn_lib_dir
|cls_thresh|float|0.9|方向分类器的得分阈值|
|cls_batch_num|int|1|方向分类器batchsize|
-- 识别模型相关
+- 文字识别模型相关
|参数名称|类型|默认参数|意义|
| :---: | :---: | :---: | :---: |
-|rec_model_dir|string|-|识别模型inference model地址|
+|rec_model_dir|string|-|文字识别模型inference model地址|
|rec_char_dict_path|string|../../ppocr/utils/ppocr_keys_v1.txt|字典文件|
-|rec_batch_num|int|6|识别模型batchsize|
-|rec_img_h|int|48|识别模型输入图像高度|
-|rec_img_w|int|320|识别模型输入图像宽度|
+|rec_batch_num|int|6|文字识别模型batchsize|
+|rec_img_h|int|48|文字识别模型输入图像高度|
+|rec_img_w|int|320|文字识别模型输入图像宽度|
+
+
+- 表格识别模型相关
+
+|参数名称|类型|默认参数|意义|
+| :---: | :---: | :---: | :---: |
+|table_model_dir|string|-|表格识别模型inference model地址|
+|table_char_dict_path|string|../../ppocr/utils/dict/table_structure_dict.txt|字典文件|
+|table_max_len|int|488|表格识别模型输入图像长边大小,最终网络输入图像大小为(table_max_len,table_max_len)|
* PaddleOCR也支持多语言的预测,更多支持的语言和模型可以参考[识别文档](../../doc/doc_ch/recognition.md)中的多语言字典与模型部分,如果希望进行多语言预测,只需将修改`rec_char_dict_path`(字典文件路径)以及`rec_model_dir`(inference模型路径)字段即可。
最终屏幕上会输出检测结果如下。
+- ocr
+
```bash
predict img: ../../doc/imgs/12.jpg
../../doc/imgs/12.jpg
@@ -353,6 +382,13 @@ predict img: ../../doc/imgs/12.jpg
The detection visualized image saved in ./output//12.jpg
```
+- table
+
+```bash
+predict img: ../../ppstructure/docs/table/table.jpg
+0 type: table, region: [0,0,371,293], res: | Methods | R | P | F | FPS | | SegLink [26] | 70.0 | 86.0 | 77.0 | 8.9 | | PixelLink [4] | 73.2 | 83.0 | 77.8 | - | | TextSnake [18] | 73.9 | 83.2 | 78.3 | 1.1 | | TextField [37] | 75.9 | 87.4 | 81.3 | 5.2 | | MSR[38] | 76.7 | 87.4 | 81.7 | - | | FTSN [3] | 77.1 | 87.6 | 82.0 | - | | LSE[30] | 81.7 | 84.2 | 82.9 | - | | CRAFT [2] | 78.2 | 88.2 | 82.9 | 8.6 | | MCN [16] | 79 | 88 | 83 | - | | ATRR[35] | 82.1 | 85.2 | 83.6 | - | | PAN [34] | 83.8 | 84.4 | 84.1 | 30.2 | | DB[12] | 79.2 | 91.5 | 84.9 | 32.0 | | DRRG [41] | 82.30 | 88.05 | 85.08 | - | | Ours (SynText) | 80.68 | 85.40 | 82.97 | 12.68 | | Ours (MLT-17) | 84.54 | 86.62 | 85.57 | 12.31 |
+```
+
## 3. FAQ
diff --git a/deploy/cpp_infer/src/main.cpp b/deploy/cpp_infer/src/main.cpp
index aa35eca3f1..66412a7b28 100644
--- a/deploy/cpp_infer/src/main.cpp
+++ b/deploy/cpp_infer/src/main.cpp
@@ -119,7 +119,7 @@ void structure(std::vector &cv_all_img_names) {
std::vector> structure_results =
engine.structure(cv_all_img_names, false, FLAGS_table);
for (int i = 0; i < cv_all_img_names.size(); i++) {
- cout << cv_all_img_names[i] << "\n";
+ cout << "predict img: " << cv_all_img_names[i] << endl;
for (int j = 0; j < structure_results[i].size(); j++) {
std::cout << j << "\ttype: " << structure_results[i][j].type
<< ", region: [";
@@ -127,7 +127,7 @@ void structure(std::vector &cv_all_img_names) {
<< structure_results[i][j].box[1] << ","
<< structure_results[i][j].box[2] << ","
<< structure_results[i][j].box[3] << "], res: ";
- if (structure_results[i][j].type == "Table") {
+ if (structure_results[i][j].type == "table") {
std::cout << structure_results[i][j].html << std::endl;
} else {
Utility::print_result(structure_results[i][j].text_res);
diff --git a/deploy/cpp_infer/src/paddlestructure.cpp b/deploy/cpp_infer/src/paddlestructure.cpp
index dbaa84fe84..1ca85a96bb 100644
--- a/deploy/cpp_infer/src/paddlestructure.cpp
+++ b/deploy/cpp_infer/src/paddlestructure.cpp
@@ -55,7 +55,7 @@ PaddleStructure::structure(std::vector cv_all_img_names,
if (layout) {
} else {
StructurePredictResult res;
- res.type = "Table";
+ res.type = "table";
res.box = std::vector(4, 0);
res.box[2] = srcimg.cols;
res.box[3] = srcimg.rows;
@@ -65,7 +65,7 @@ PaddleStructure::structure(std::vector cv_all_img_names,
for (int i = 0; i < structure_result.size(); i++) {
// crop image
roi_img = Utility::crop_image(srcimg, structure_result[i].box);
- if (structure_result[i].type == "Table") {
+ if (structure_result[i].type == "table") {
this->table(roi_img, structure_result[i], time_info_table,
time_info_det, time_info_rec, time_info_cls);
}
From 38b3ca47a9650f476879172051b01a935ff88dae Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Wed, 10 Aug 2022 07:07:32 +0000
Subject: [PATCH 10/49] avoid cnflic
---
ppstructure/utility.py | 1 -
1 file changed, 1 deletion(-)
diff --git a/ppstructure/utility.py b/ppstructure/utility.py
index f5388fabf2..767c5704fa 100644
--- a/ppstructure/utility.py
+++ b/ppstructure/utility.py
@@ -32,7 +32,6 @@ def init_args():
type=str,
default="../ppocr/utils/dict/table_structure_dict.txt")
# params for layout
- parser.add_argument("--layout_model_dir", type=str)
parser.add_argument(
"--layout_path_model",
type=str,
From 1e27820f59c0234d5db8c591f7701bf33be3d346 Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Wed, 10 Aug 2022 10:52:34 +0000
Subject: [PATCH 11/49] add merge flag
---
ppstructure/table/predict_structure.py | 4 ++-
ppstructure/table/predict_table.py | 38 ++++++++++++++++++++++++--
ppstructure/utility.py | 2 ++
3 files changed, 41 insertions(+), 3 deletions(-)
diff --git a/ppstructure/table/predict_structure.py b/ppstructure/table/predict_structure.py
index 01d4675943..c4a816fd87 100755
--- a/ppstructure/table/predict_structure.py
+++ b/ppstructure/table/predict_structure.py
@@ -73,12 +73,14 @@ class TableStructurer(object):
postprocess_params = {
'name': 'TableLabelDecode',
"character_dict_path": args.table_char_dict_path,
+ 'merge_no_span_structure': args.merge_no_span_structure
}
else:
postprocess_params = {
'name': 'TableMasterLabelDecode',
"character_dict_path": args.table_char_dict_path,
- 'box_shape': 'pad'
+ 'box_shape': 'pad',
+ 'merge_no_span_structure': args.merge_no_span_structure
}
self.preprocess_op = create_operators(pre_process_list)
diff --git a/ppstructure/table/predict_table.py b/ppstructure/table/predict_table.py
index b0c7ef589f..35ce8890cf 100644
--- a/ppstructure/table/predict_table.py
+++ b/ppstructure/table/predict_table.py
@@ -101,6 +101,7 @@ class TableSystem(object):
start = time.time()
structure_res, elapse = self._structure(copy.deepcopy(img))
+ result['cell_bbox'] = structure_res[1]
time_dict['table'] = elapse
dt_boxes, rec_res, det_elapse, rec_elapse = self._ocr(
@@ -175,8 +176,23 @@ def main(args):
image_file_list = image_file_list[args.process_id::args.total_process_num]
os.makedirs(args.output, exist_ok=True)
- text_sys = TableSystem(args)
+ table_sys = TableSystem(args)
img_num = len(image_file_list)
+
+ f_html = open(
+ os.path.join(args.output, 'show.html'), mode='w', encoding='utf-8')
+ f_html.write('\n\n')
+ f_html.write('\n')
+ f_html.write(
+ ""
+ )
+ f_html.write("\n")
+ f_html.write('| img name\n')
+ f_html.write(' | ori image | ')
+ f_html.write('table html | ')
+ f_html.write('cell box | ')
+ f_html.write(" \n")
+
for i, image_file in enumerate(image_file_list):
logger.info("[{}/{}] {}".format(i, img_num, image_file))
img, flag = check_and_read_gif(image_file)
@@ -188,13 +204,31 @@ def main(args):
logger.error("error in loading image:{}".format(image_file))
continue
starttime = time.time()
- pred_res, _ = text_sys(img)
+ pred_res, _ = table_sys(img)
pred_html = pred_res['html']
logger.info(pred_html)
to_excel(pred_html, excel_path)
logger.info('excel saved to {}'.format(excel_path))
elapse = time.time() - starttime
logger.info("Predict time : {:.3f}s".format(elapse))
+
+ # img = predict_strture.draw_rectangle(image_file, pred_res['cell_bbox'], use_xywh)
+ img = utility.draw_boxes(cv2.imread(image_file), pred_res['cell_bbox'])
+ img_save_path = os.path.join(args.output, os.path.basename(image_file))
+ cv2.imwrite(img_save_path, img)
+
+ f_html.write("\n")
+ f_html.write(f' {os.path.basename(image_file)} \n')
+ f_html.write(f' |  | \n')
+ f_html.write('' + pred_html.replace(
+ '', '') +
+ ' | \n')
+ f_html.write(
+ f'}) | \n')
+ f_html.write(" \n")
+ f_html.write(" \n")
+ f_html.close()
+
if args.benchmark:
text_sys.autolog.report()
diff --git a/ppstructure/utility.py b/ppstructure/utility.py
index 767c5704fa..1c77cecd5c 100644
--- a/ppstructure/utility.py
+++ b/ppstructure/utility.py
@@ -27,6 +27,8 @@ def init_args():
parser.add_argument("--table_max_len", type=int, default=488)
parser.add_argument("--table_algorithm", type=str, default='TableAttn')
parser.add_argument("--table_model_dir", type=str)
+ parser.add_argument(
+ "--merge_no_span_structure", type=str2bool, default=False)
parser.add_argument(
"--table_char_dict_path",
type=str,
From de712e2762df23bff12152857a6667e052f7049f Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Wed, 10 Aug 2022 11:01:03 +0000
Subject: [PATCH 12/49] update ch doc
---
ppstructure/table/README_ch.md | 27 +++++++++++++--------------
1 file changed, 13 insertions(+), 14 deletions(-)
diff --git a/ppstructure/table/README_ch.md b/ppstructure/table/README_ch.md
index a0a64d6b7e..21fb7960cc 100644
--- a/ppstructure/table/README_ch.md
+++ b/ppstructure/table/README_ch.md
@@ -40,7 +40,8 @@
|算法|[TEDS(Tree-Edit-Distance-based Similarity)](https://github.com/ibm-aur-nlp/PubTabNet/tree/master/src)|
| --- | --- |
| EDD[2] | 88.3 |
-| Ours | 93.32 |
+| TableRec-RARE(ours) | 93.32 |
+| SLANet(ours) | 94.98 |
## 3. 使用
@@ -63,7 +64,7 @@ cd ..
# 执行预测
python3 table/predict_table.py --det_model_dir=inference/en_ppocr_mobile_v2.0_table_det_infer --rec_model_dir=inference/en_ppocr_mobile_v2.0_table_rec_infer --table_model_dir=inference/en_ppocr_mobile_v2.0_table_structure_infer --image_dir=./docs/table/table.jpg --rec_char_dict_path=../ppocr/utils/dict/table_dict.txt --table_char_dict_path=../ppocr/utils/dict/table_structure_dict.txt --det_limit_side_len=736 --det_limit_type=min --output ./output/table
```
-运行完成后,每张图片的excel表格会保存到output字段指定的目录下
+运行完成后,每张图片的excel表格会保存到output字段指定的目录下,同时在该目录下回生产一个html文件,用于可视化查看单元格坐标和识别的表格。
note: 上述模型是在 PubLayNet 数据集上训练的表格识别模型,仅支持英文扫描场景,如需识别其他场景需要自己训练模型后替换 `det_model_dir`,`rec_model_dir`,`table_model_dir`三个字段即可。
@@ -101,26 +102,24 @@ python3 tools/train.py -c configs/table/table_mv3.yml -o Global.checkpoints=./yo
### 3.3 评估
表格使用 [TEDS(Tree-Edit-Distance-based Similarity)](https://github.com/ibm-aur-nlp/PubTabNet/tree/master/src) 作为模型的评估指标。在进行模型评估之前,需要将pipeline中的三个模型分别导出为inference模型(我们已经提供好),还需要准备评估的gt, gt示例如下:
-```json
-{"PMC4289340_004_00.png": [
- ["", "", "", "", "", "| ", " | ", "", " | ", "", " | ", " ", "", "", "", "| ", " | ", "", " | ", "", " | ", " ", "", " ", "", ""],
- [[1, 4, 29, 13], [137, 4, 161, 13], [215, 4, 236, 13], [1, 17, 30, 27], [137, 17, 147, 27], [215, 17, 225, 27]],
- [["", "F", "e", "a", "t", "u", "r", "e", ""], ["", "G", "b", "3", " ", "+", ""], ["", "G", "b", "3", " ", "-", ""], ["", "P", "a", "t", "i", "e", "n", "t", "s", ""], ["6", "2"], ["4", "5"]]
-]}
+```txt
+PMC5755158_010_01.png | Weaning | Week 15 | Off-test | | Weaning | – | – | – | | Week 15 | – | 0.17 ± 0.08 | 0.16 ± 0.03 | | Off-test | – | 0.80 ± 0.24 | 0.19 ± 0.09 |
+```
+gt每一行都由文件名和表格的html字符串组成,文件名和表格的html字符串之间使用`\t`分隔。
+
+也可使用如下命令,由标注文件生成评估的gt文件:
+```python
+python3 ppstructure/table/convert_label2html.py --ori_gt_path /path/to/your_label_file --save_path /path/to/save_file
```
-json 中,key为图片名,value为对应的gt,gt是一个由三个item组成的list,每个item分别为
-1. 表格结构的html字符串list
-2. 每个cell的坐标 (不包括cell里文字为空的)
-3. 每个cell里的文字信息 (不包括cell里文字为空的)
准备完成后使用如下命令进行评估,评估完成后会输出teds指标。
```python
cd PaddleOCR/ppstructure
-python3 table/eval_table.py --det_model_dir=path/to/det_model_dir --rec_model_dir=path/to/rec_model_dir --table_model_dir=path/to/table_model_dir --image_dir=../doc/table/1.png --rec_char_dict_path=../ppocr/utils/dict/table_dict.txt --table_char_dict_path=../ppocr/utils/dict/table_structure_dict.txt --det_limit_side_len=736 --det_limit_type=min --gt_path=path/to/gt.json
+python3 table/eval_table.py --det_model_dir=path/to/det_model_dir --rec_model_dir=path/to/rec_model_dir --table_model_dir=path/to/table_model_dir --image_dir=../doc/table/1.png --rec_char_dict_path=../ppocr/utils/dict/table_dict.txt --table_char_dict_path=../ppocr/utils/dict/table_structure_dict.txt --det_limit_side_len=736 --det_limit_type=min --gt_path=path/to/gt.txt
```
如使用PubLatNet评估数据集,将会输出
```bash
-teds: 93.32
+teds: 94.98
```
From 4ac17fca551cc3b2085351842f15c88405cc8d57 Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Wed, 10 Aug 2022 11:04:06 +0000
Subject: [PATCH 13/49] update train.py
---
tools/train.py | 2 --
1 file changed, 2 deletions(-)
diff --git a/tools/train.py b/tools/train.py
index ddd7c312a9..0c881ecae8 100755
--- a/tools/train.py
+++ b/tools/train.py
@@ -123,8 +123,6 @@ def main(config, device, logger, vdl_writer):
if use_sync_bn:
model = paddle.nn.SyncBatchNorm.convert_sync_batchnorm(model)
logger.info('convert_sync_batchnorm')
- if config['Global']['distributed']:
- model = paddle.DataParallel(model)
model = apply_to_static(model, config, logger)
From 73ca6c2e7f04f1a2048222e06245ffba17745f09 Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Wed, 10 Aug 2022 14:15:52 +0000
Subject: [PATCH 14/49] add PP-Structurev2 to hubserving
---
deploy/hubserving/readme.md | 3 +-
deploy/hubserving/readme_en.md | 3 +-
deploy/hubserving/structure_system/module.py | 4 +--
deploy/hubserving/structure_system/params.py | 6 ++--
deploy/hubserving/structure_table/module.py | 6 ++--
doc/doc_en/quickstart_en.md | 20 ++++--------
ppstructure/docs/installation.md | 12 ++-----
ppstructure/docs/quickstart.md | 34 +++++++++-----------
ppstructure/docs/quickstart_en.md | 34 +++++++++-----------
ppstructure/predict_system.py | 5 +--
ppstructure/table/predict_table.py | 2 +-
ppstructure/utility.py | 7 ++--
12 files changed, 63 insertions(+), 73 deletions(-)
diff --git a/deploy/hubserving/readme.md b/deploy/hubserving/readme.md
index 183a25912c..2264e6eaa4 100755
--- a/deploy/hubserving/readme.md
+++ b/deploy/hubserving/readme.md
@@ -59,6 +59,7 @@ pip3 install paddlehub==2.1.0 --upgrade -i https://mirror.baidu.com/pypi/simple
检测模型:./inference/ch_PP-OCRv3_det_infer/
识别模型:./inference/ch_PP-OCRv3_rec_infer/
方向分类器:./inference/ch_ppocr_mobile_v2.0_cls_infer/
+版面分析模型:
表格结构识别模型:./inference/en_ppocr_mobile_v2.0_table_structure_infer/
```
@@ -172,7 +173,7 @@ hub serving start -c deploy/hubserving/ocr_system/config.json
## 3. 发送预测请求
配置好服务端,可使用以下命令发送预测请求,获取预测结果:
-```python tools/test_hubserving.py server_url image_path```
+```python tools/test_hubserving.py --server_url=server_url --image_dir=image_path```
需要给脚本传递2个参数:
- **server_url**:服务地址,格式为
diff --git a/deploy/hubserving/readme_en.md b/deploy/hubserving/readme_en.md
index 27eccbb5e9..6463fcb54b 100755
--- a/deploy/hubserving/readme_en.md
+++ b/deploy/hubserving/readme_en.md
@@ -61,6 +61,7 @@ Before installing the service module, you need to prepare the inference model an
text detection model: ./inference/ch_PP-OCRv3_det_infer/
text recognition model: ./inference/ch_PP-OCRv3_rec_infer/
text angle classifier: ./inference/ch_ppocr_mobile_v2.0_cls_infer/
+layout parse model:
tanle recognition: ./inference/en_ppocr_mobile_v2.0_table_structure_infer/
```
@@ -177,7 +178,7 @@ hub serving start -c deploy/hubserving/ocr_system/config.json
## 3. Send prediction requests
After the service starts, you can use the following command to send a prediction request to obtain the prediction result:
```shell
-python tools/test_hubserving.py server_url image_path
+python tools/test_hubserving.py --server_url=server_url --image_dir=image_path
```
Two parameters need to be passed to the script:
diff --git a/deploy/hubserving/structure_system/module.py b/deploy/hubserving/structure_system/module.py
index 92846edc66..61c93bb146 100644
--- a/deploy/hubserving/structure_system/module.py
+++ b/deploy/hubserving/structure_system/module.py
@@ -119,7 +119,7 @@ class StructureSystem(hub.Module):
all_results.append([])
continue
starttime = time.time()
- res = self.table_sys(img)
+ res, _ = self.table_sys(img)
elapse = time.time() - starttime
logger.info("Predict time: {}".format(elapse))
@@ -144,6 +144,6 @@ class StructureSystem(hub.Module):
if __name__ == '__main__':
structure_system = StructureSystem()
structure_system._initialize()
- image_path = ['./doc/table/1.png']
+ image_path = ['./ppstructure/docs/table/1.png']
res = structure_system.predict(paths=image_path)
print(res)
diff --git a/deploy/hubserving/structure_system/params.py b/deploy/hubserving/structure_system/params.py
index 3cc6a2794f..fe691fbc2d 100755
--- a/deploy/hubserving/structure_system/params.py
+++ b/deploy/hubserving/structure_system/params.py
@@ -23,8 +23,10 @@ def read_params():
cfg = table_read_params()
# params for layout parser model
- cfg.layout_path_model = 'lp://PubLayNet/ppyolov2_r50vd_dcn_365e_publaynet/config'
- cfg.layout_label_map = None
+ cfg.layout_model_dir = ''
+ cfg.layout_dict_path = './ppocr/utils/dict/layout_publaynet_dict.txt'
+ cfg.layout_score_threshold = 0.5
+ cfg.layout_nms_threshold = 0.5
cfg.mode = 'structure'
cfg.output = './output'
diff --git a/deploy/hubserving/structure_table/module.py b/deploy/hubserving/structure_table/module.py
index 00393daa03..b4432b2d7b 100644
--- a/deploy/hubserving/structure_table/module.py
+++ b/deploy/hubserving/structure_table/module.py
@@ -118,11 +118,11 @@ class TableSystem(hub.Module):
all_results.append([])
continue
starttime = time.time()
- pred_html = self.table_sys(img)
+ res, _ = self.table_sys(img)
elapse = time.time() - starttime
logger.info("Predict time: {}".format(elapse))
- all_results.append({'html': pred_html})
+ all_results.append({'html': res['html']})
return all_results
@serving
@@ -138,6 +138,6 @@ class TableSystem(hub.Module):
if __name__ == '__main__':
table_system = TableSystem()
table_system._initialize()
- image_path = ['./doc/table/table.jpg']
+ image_path = ['./ppstructure/docs/table/table.jpg']
res = table_system.predict(paths=image_path)
print(res)
diff --git a/doc/doc_en/quickstart_en.md b/doc/doc_en/quickstart_en.md
index c678dc4762..9e1de839ff 100644
--- a/doc/doc_en/quickstart_en.md
+++ b/doc/doc_en/quickstart_en.md
@@ -3,14 +3,14 @@
**Note:** This tutorial mainly introduces the usage of PP-OCR series models, please refer to [PP-Structure Quick Start](../../ppstructure/docs/quickstart_en.md) for the quick use of document analysis related functions.
- [1. Installation](#1-installation)
- - [1.1 Install PaddlePaddle](#11-install-paddlepaddle)
- - [1.2 Install PaddleOCR Whl Package](#12-install-paddleocr-whl-package)
+ - [1.1 Install PaddlePaddle](#11-install-paddlepaddle)
+ - [1.2 Install PaddleOCR Whl Package](#12-install-paddleocr-whl-package)
- [2. Easy-to-Use](#2-easy-to-use)
- - [2.1 Use by Command Line](#21-use-by-command-line)
- - [2.1.1 Chinese and English Model](#211-chinese-and-english-model)
- - [2.1.2 Multi-language Model](#212-multi-language-model)
- - [2.2 Use by Code](#22-use-by-code)
- - [2.2.1 Chinese & English Model and Multilingual Model](#221-chinese--english-model-and-multilingual-model)
+ - [2.1 Use by Command Line](#21-use-by-command-line)
+ - [2.1.1 Chinese and English Model](#211-chinese-and-english-model)
+ - [2.1.2 Multi-language Model](#212-multi-language-model)
+ - [2.2 Use by Code](#22-use-by-code)
+ - [2.2.1 Chinese & English Model and Multilingual Model](#221-chinese--english-model-and-multilingual-model)
- [3. Summary](#3-summary)
@@ -51,12 +51,6 @@ pip install "paddleocr>=2.0.1" # Recommend to use version 2.0.1+
Reference: [Solve shapely installation on windows](https://stackoverflow.com/questions/44398265/install-shapely-oserror-winerror-126-the-specified-module-could-not-be-found)
-- **For layout analysis users**, run the following command to install **Layout-Parser**
-
- ```bash
- pip3 install -U https://paddleocr.bj.bcebos.com/whl/layoutparser-0.0.0-py3-none-any.whl
- ```
-
## 2. Easy-to-Use
diff --git a/ppstructure/docs/installation.md b/ppstructure/docs/installation.md
index 155baf29de..3f564cb2dd 100644
--- a/ppstructure/docs/installation.md
+++ b/ppstructure/docs/installation.md
@@ -1,8 +1,7 @@
- [快速安装](#快速安装)
- [1. PaddlePaddle 和 PaddleOCR](#1-paddlepaddle-和-paddleocr)
- [2. 安装其他依赖](#2-安装其他依赖)
- - [2.1 版面分析所需 Layout-Parser](#21-版面分析所需--layout-parser)
- - [2.2 VQA所需依赖](#22--vqa所需依赖)
+ - [2.1 VQA所需依赖](#21--vqa所需依赖)
# 快速安装
@@ -12,14 +11,7 @@
## 2. 安装其他依赖
-### 2.1 版面分析所需 Layout-Parser
-
-Layout-Parser 可通过如下命令安装
-
-```bash
-pip3 install -U https://paddleocr.bj.bcebos.com/whl/layoutparser-0.0.0-py3-none-any.whl
-```
-### 2.2 VQA所需依赖
+### 2.1 VQA所需依赖
* paddleocr
```bash
diff --git a/ppstructure/docs/quickstart.md b/ppstructure/docs/quickstart.md
index 31e5941624..d206d1d521 100644
--- a/ppstructure/docs/quickstart.md
+++ b/ppstructure/docs/quickstart.md
@@ -1,21 +1,21 @@
# PP-Structure 快速开始
-- [1. 安装依赖包](#1)
-- [2. 便捷使用](#2)
- - [2.1 命令行使用](#21)
- - [2.1.1 版面分析+表格识别](#211)
- - [2.1.2 版面分析](#212)
- - [2.1.3 表格识别](#213)
- - [2.1.4 DocVQA](#214)
- - [2.2 代码使用](#22)
- - [2.2.1 版面分析+表格识别](#221)
- - [2.2.2 版面分析](#222)
- - [2.2.3 表格识别](#223)
- - [2.2.4 DocVQA](#224)
- - [2.3 返回结果说明](#23)
- - [2.3.1 版面分析+表格识别](#231)
- - [2.3.2 DocVQA](#232)
- - [2.4 参数说明](#24)
+- [1. 安装依赖包](#1-安装依赖包)
+- [2. 便捷使用](#2-便捷使用)
+ - [2.1 命令行使用](#21-命令行使用)
+ - [2.1.1 版面分析+表格识别](#211-版面分析表格识别)
+ - [2.1.2 版面分析](#212-版面分析)
+ - [2.1.3 表格识别](#213-表格识别)
+ - [2.1.4 DocVQA](#214-docvqa)
+ - [2.2 代码使用](#22-代码使用)
+ - [2.2.1 版面分析+表格识别](#221-版面分析表格识别)
+ - [2.2.2 版面分析](#222-版面分析)
+ - [2.2.3 表格识别](#223-表格识别)
+ - [2.2.4 DocVQA](#224-docvqa)
+ - [2.3 返回结果说明](#23-返回结果说明)
+ - [2.3.1 版面分析+表格识别](#231-版面分析表格识别)
+ - [2.3.2 DocVQA](#232-docvqa)
+ - [2.4 参数说明](#24-参数说明)
@@ -24,8 +24,6 @@
```bash
# 安装 paddleocr,推荐使用2.5+版本
pip3 install "paddleocr>=2.5"
-# 安装 版面分析依赖包layoutparser(如不需要版面分析功能,可跳过)
-pip3 install -U https://paddleocr.bj.bcebos.com/whl/layoutparser-0.0.0-py3-none-any.whl
# 安装 DocVQA依赖包paddlenlp(如不需要DocVQA功能,可跳过)
pip install paddlenlp
diff --git a/ppstructure/docs/quickstart_en.md b/ppstructure/docs/quickstart_en.md
index 1f78b43ea3..98d8d2fc3f 100644
--- a/ppstructure/docs/quickstart_en.md
+++ b/ppstructure/docs/quickstart_en.md
@@ -1,21 +1,21 @@
# PP-Structure Quick Start
-- [1. Install package](#1)
-- [2. Use](#2)
- - [2.1 Use by command line](#21)
- - [2.1.1 layout analysis + table recognition](#211)
- - [2.1.2 layout analysis](#212)
- - [2.1.3 table recognition](#213)
- - [2.1.4 DocVQA](#214)
- - [2.2 Use by code](#22)
- - [2.2.1 layout analysis + table recognition](#221)
- - [2.2.2 layout analysis](#222)
- - [2.2.3 table recognition](#223)
- - [2.2.4 DocVQA](#224)
- - [2.3 Result description](#23)
- - [2.3.1 layout analysis + table recognition](#231)
- - [2.3.2 DocVQA](#232)
- - [2.4 Parameter Description](#24)
+- [1. Install package](#1-install-package)
+- [2. Use](#2-use)
+ - [2.1 Use by command line](#21-use-by-command-line)
+ - [2.1.1 layout analysis + table recognition](#211-layout-analysis--table-recognition)
+ - [2.1.2 layout analysis](#212-layout-analysis)
+ - [2.1.3 table recognition](#213-table-recognition)
+ - [2.1.4 DocVQA](#214-docvqa)
+ - [2.2 Use by code](#22-use-by-code)
+ - [2.2.1 layout analysis + table recognition](#221-layout-analysis--table-recognition)
+ - [2.2.2 layout analysis](#222-layout-analysis)
+ - [2.2.3 table recognition](#223-table-recognition)
+ - [2.2.4 DocVQA](#224-docvqa)
+ - [2.3 Result description](#23-result-description)
+ - [2.3.1 layout analysis + table recognition](#231-layout-analysis--table-recognition)
+ - [2.3.2 DocVQA](#232-docvqa)
+ - [2.4 Parameter Description](#24-parameter-description)
@@ -24,8 +24,6 @@
```bash
# Install paddleocr, version 2.5+ is recommended
pip3 install "paddleocr>=2.5"
-# Install layoutparser (if you do not use the layout analysis, you can skip it)
-pip3 install -U https://paddleocr.bj.bcebos.com/whl/layoutparser-0.0.0-py3-none-any.whl
# Install the DocVQA dependency package paddlenlp (if you do not use the DocVQA, you can skip it)
pip install paddlenlp
diff --git a/ppstructure/predict_system.py b/ppstructure/predict_system.py
index 075d914461..608f4d2fb3 100644
--- a/ppstructure/predict_system.py
+++ b/ppstructure/predict_system.py
@@ -43,6 +43,7 @@ logger = get_logger()
class StructureSystem(object):
def __init__(self, args):
self.mode = args.mode
+ self.recovery = args.recovery
if self.mode == 'structure':
if not args.show_log:
logger.setLevel(logging.INFO)
@@ -110,7 +111,7 @@ class StructureSystem(object):
time_dict['rec'] += table_time_dict['rec']
else:
if self.text_system is not None:
- if args.recovery:
+ if self.recovery:
wht_im = np.ones(ori_im.shape, dtype=ori_im.dtype)
wht_im[y1:y2, x1:x2, :] = roi_img
filter_boxes, filter_rec_res, ocr_time_dict = self.text_system(
@@ -133,7 +134,7 @@ class StructureSystem(object):
for token in style_token:
if token in rec_str:
rec_str = rec_str.replace(token, '')
- if not args.recovery:
+ if not self.recovery:
box += [x1, y1]
res.append({
'text': rec_str,
diff --git a/ppstructure/table/predict_table.py b/ppstructure/table/predict_table.py
index 35ce8890cf..f580213753 100644
--- a/ppstructure/table/predict_table.py
+++ b/ppstructure/table/predict_table.py
@@ -101,7 +101,7 @@ class TableSystem(object):
start = time.time()
structure_res, elapse = self._structure(copy.deepcopy(img))
- result['cell_bbox'] = structure_res[1]
+ result['cell_bbox'] = structure_res[1].tolist()
time_dict['table'] = elapse
dt_boxes, rec_res, det_elapse, rec_elapse = self._ocr(
diff --git a/ppstructure/utility.py b/ppstructure/utility.py
index 390736cda9..597d978516 100644
--- a/ppstructure/utility.py
+++ b/ppstructure/utility.py
@@ -38,14 +38,17 @@ def init_args():
parser.add_argument(
"--layout_dict_path",
type=str,
- default="../ppocr/utils/dict/layout_pubalynet_dict.txt")
+ default="../ppocr/utils/dict/layout_publaynet_dict.txt")
parser.add_argument(
"--layout_score_threshold",
type=float,
default=0.5,
help="Threshold of score.")
parser.add_argument(
- "--layout_nms_threshold", type=float, default=0.5, help="Threshold of nms.")
+ "--layout_nms_threshold",
+ type=float,
+ default=0.5,
+ help="Threshold of nms.")
# params for vqa
parser.add_argument("--vqa_algorithm", type=str, default='LayoutXLM')
parser.add_argument("--ser_model_dir", type=str)
From 731688c2dd359a289081b4b9ef8626217dd3a19c Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Wed, 10 Aug 2022 14:51:57 +0000
Subject: [PATCH 15/49] add PP-Structurev2 to whl
---
paddleocr.py | 50 ++++++++++++++++++++++++++++++++++++++++----------
1 file changed, 40 insertions(+), 10 deletions(-)
diff --git a/paddleocr.py b/paddleocr.py
index 470dc60da3..9a9958abef 100644
--- a/paddleocr.py
+++ b/paddleocr.py
@@ -47,14 +47,14 @@ __all__ = [
]
SUPPORT_DET_MODEL = ['DB']
-VERSION = '2.5.0.3'
+VERSION = '2.6'
SUPPORT_REC_MODEL = ['CRNN', 'SVTR_LCNet']
BASE_DIR = os.path.expanduser("~/.paddleocr/")
DEFAULT_OCR_MODEL_VERSION = 'PP-OCRv3'
SUPPORT_OCR_MODEL_VERSION = ['PP-OCR', 'PP-OCRv2', 'PP-OCRv3']
-DEFAULT_STRUCTURE_MODEL_VERSION = 'PP-STRUCTURE'
-SUPPORT_STRUCTURE_MODEL_VERSION = ['PP-STRUCTURE']
+DEFAULT_STRUCTURE_MODEL_VERSION = 'PP-Structurev2'
+SUPPORT_STRUCTURE_MODEL_VERSION = ['PP-Structure', 'PP-Structurev2']
MODEL_URLS = {
'OCR': {
'PP-OCRv3': {
@@ -263,7 +263,7 @@ MODEL_URLS = {
}
},
'STRUCTURE': {
- 'PP-STRUCTURE': {
+ 'PP-Structure': {
'table': {
'en': {
'url':
@@ -271,6 +271,24 @@ MODEL_URLS = {
'dict_path': 'ppocr/utils/dict/table_structure_dict.txt'
}
}
+ },
+ 'PP-Structurev2': {
+ 'table': {
+ 'en': {
+ 'url': '',
+ 'dict_path': 'ppocr/utils/dict/table_structure_dict.txt'
+ },
+ 'ch': {
+ 'url': '',
+ 'dict_path': 'ppocr/utils/dict/table_structure_dict.txt'
+ }
+ },
+ 'layout': {
+ 'ch': {
+ 'url': '',
+ 'dict_path': 'ppocr/utils/dict/layout_publaynet_dict.txt'
+ }
+ }
}
}
}
@@ -298,12 +316,15 @@ def parse_args(mMain=True):
"--structure_version",
type=str,
choices=SUPPORT_STRUCTURE_MODEL_VERSION,
- default='PP-STRUCTURE',
+ default='PP-Structure',
help='Model version, the current model support list is as follows:'
- ' 1. STRUCTURE Support en table structure model.')
+ ' 1. PP-Structure Support en table structure model.'
+ ' 2. PP-Structure Support ch and en table structure model.')
for action in parser._actions:
- if action.dest in ['rec_char_dict_path', 'table_char_dict_path']:
+ if action.dest in [
+ 'rec_char_dict_path', 'table_char_dict_path', 'layout_dict_path'
+ ]:
action.default = None
if mMain:
return parser.parse_args()
@@ -477,7 +498,7 @@ class PaddleOCR(predict_system.TextSystem):
if isinstance(img, np.ndarray) and len(img.shape) == 2:
img = cv2.cvtColor(img, cv2.COLOR_GRAY2BGR)
if det and rec:
- dt_boxes, rec_res = self.__call__(img, cls)
+ dt_boxes, rec_res, _ = self.__call__(img, cls)
return [[box.tolist(), res] for box, res in zip(dt_boxes, rec_res)]
elif det and not rec:
dt_boxes, elapse = self.text_detector(img)
@@ -520,14 +541,20 @@ class PPStructure(StructureSystem):
params.rec_model_dir,
os.path.join(BASE_DIR, 'whl', 'rec', lang), rec_model_config['url'])
table_model_config = get_model_config(
- 'STRUCTURE', params.structure_version, 'table', 'en')
+ 'STRUCTURE', params.structure_version, 'table', 'ch')
params.table_model_dir, table_url = confirm_model_dir_url(
params.table_model_dir,
os.path.join(BASE_DIR, 'whl', 'table'), table_model_config['url'])
+ layout_model_config = get_model_config(
+ 'STRUCTURE', params.structure_version, 'layout', 'ch')
+ params.layout_model_dir, layout_url = confirm_model_dir_url(
+ params.layout_model_dir,
+ os.path.join(BASE_DIR, 'whl', 'layout'), layout_model_config['url'])
# download model
maybe_download(params.det_model_dir, det_url)
maybe_download(params.rec_model_dir, rec_url)
maybe_download(params.table_model_dir, table_url)
+ maybe_download(params.layout_model_dir, layout_url)
if params.rec_char_dict_path is None:
params.rec_char_dict_path = str(
@@ -535,6 +562,9 @@ class PPStructure(StructureSystem):
if params.table_char_dict_path is None:
params.table_char_dict_path = str(
Path(__file__).parent / table_model_config['dict_path'])
+ if params.layout_dict_path is None:
+ params.layout_dict_path = str(
+ Path(__file__).parent / layout_model_config['dict_path'])
logger.debug(params)
super().__init__(params)
@@ -557,7 +587,7 @@ class PPStructure(StructureSystem):
if isinstance(img, np.ndarray) and len(img.shape) == 2:
img = cv2.cvtColor(img, cv2.COLOR_GRAY2BGR)
- res = super().__call__(img, return_ocr_result_in_table)
+ res, _ = super().__call__(img, return_ocr_result_in_table)
return res
From c2c43bb1bc757bc5ea565ab59a23294f7a172517 Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Wed, 10 Aug 2022 14:58:08 +0000
Subject: [PATCH 16/49] rename SLANetLoss to SLALoss
---
configs/table/SLANet.yml | 2 +-
ppocr/losses/__init__.py | 4 ++--
ppocr/losses/table_att_loss.py | 4 ++--
3 files changed, 5 insertions(+), 5 deletions(-)
diff --git a/configs/table/SLANet.yml b/configs/table/SLANet.yml
index acf8d03045..4858c71c28 100644
--- a/configs/table/SLANet.yml
+++ b/configs/table/SLANet.yml
@@ -54,7 +54,7 @@ Architecture:
loc_reg_num: &loc_reg_num 4
Loss:
- name: SLANetLoss
+ name: SLALoss
structure_weight: 1.0
loc_weight: 2.0
loc_loss: smooth_l1
diff --git a/ppocr/losses/__init__.py b/ppocr/losses/__init__.py
index 8f3adfccd4..3ac766da92 100755
--- a/ppocr/losses/__init__.py
+++ b/ppocr/losses/__init__.py
@@ -52,7 +52,7 @@ from .basic_loss import DistanceLoss
from .combined_loss import CombinedLoss
# table loss
-from .table_att_loss import TableAttentionLoss, SLANetLoss
+from .table_att_loss import TableAttentionLoss, SLALoss
from .table_master_loss import TableMasterLoss
# vqa token loss
from .vqa_token_layoutlm_loss import VQASerTokenLayoutLMLoss
@@ -64,7 +64,7 @@ def build_loss(config):
'ClsLoss', 'AttentionLoss', 'SRNLoss', 'PGLoss', 'CombinedLoss',
'CELoss', 'TableAttentionLoss', 'SARLoss', 'AsterLoss', 'SDMGRLoss',
'VQASerTokenLayoutLMLoss', 'LossFromOutput', 'PRENLoss', 'MultiLoss',
- 'TableMasterLoss', 'SPINAttentionLoss', 'VLLoss', 'SLANetLoss'
+ 'TableMasterLoss', 'SPINAttentionLoss', 'VLLoss', 'SLALoss'
]
config = copy.deepcopy(config)
module_name = config.pop('name')
diff --git a/ppocr/losses/table_att_loss.py b/ppocr/losses/table_att_loss.py
index d97715d541..f1771847b4 100644
--- a/ppocr/losses/table_att_loss.py
+++ b/ppocr/losses/table_att_loss.py
@@ -55,9 +55,9 @@ class TableAttentionLoss(nn.Layer):
}
-class SLANetLoss(nn.Layer):
+class SLALoss(nn.Layer):
def __init__(self, structure_weight, loc_weight, loc_loss='mse', **kwargs):
- super(SLANetLoss, self).__init__()
+ super(SLALoss, self).__init__()
self.loss_func = nn.CrossEntropyLoss(weight=None, reduction='mean')
self.structure_weight = structure_weight
self.loc_weight = loc_weight
From 73c77ff79d418e916fbfa582502f25affb4f0434 Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Thu, 11 Aug 2022 10:56:19 +0000
Subject: [PATCH 17/49] add image_orientation and update quickstart
---
ppocr/metrics/table_metric.py | 18 ++++-
ppstructure/docs/quickstart.md | 121 +++++++++++++++++++----------
ppstructure/docs/quickstart_en.md | 122 ++++++++++++++++++++----------
ppstructure/predict_system.py | 32 ++++++--
ppstructure/utility.py | 5 ++
5 files changed, 213 insertions(+), 85 deletions(-)
diff --git a/ppocr/metrics/table_metric.py b/ppocr/metrics/table_metric.py
index 43dc1d7617..c0b247efa6 100644
--- a/ppocr/metrics/table_metric.py
+++ b/ppocr/metrics/table_metric.py
@@ -16,9 +16,14 @@ from ppocr.metrics.det_metric import DetMetric
class TableStructureMetric(object):
- def __init__(self, main_indicator='acc', eps=1e-6, **kwargs):
+ def __init__(self,
+ main_indicator='acc',
+ eps=1e-6,
+ del_thead_tbody=False,
+ **kwargs):
self.main_indicator = main_indicator
self.eps = eps
+ self.del_thead_tbody = del_thead_tbody
self.reset()
def __call__(self, pred_label, batch=None, *args, **kwargs):
@@ -31,6 +36,13 @@ class TableStructureMetric(object):
gt_structure_batch_list):
pred_str = ''.join(pred)
target_str = ''.join(target)
+ if self.del_thead_tbody:
+ pred_str = pred_str.replace('', '').replace(
+ '', '').replace('', '').replace('',
+ '')
+ target_str = target_str.replace('', '').replace(
+ '', '').replace('', '').replace('',
+ '')
if pred_str == target_str:
correct_num += 1
all_num += 1
@@ -60,6 +72,7 @@ class TableMetric(object):
main_indicator='acc',
compute_bbox_metric=False,
box_format='xyxy',
+ del_thead_tbody=False,
**kwargs):
"""
@@ -67,7 +80,8 @@ class TableMetric(object):
@param main_matric: main_matric for save best_model
@param kwargs:
"""
- self.structure_metric = TableStructureMetric()
+ self.structure_metric = TableStructureMetric(
+ del_thead_tbody=del_thead_tbody)
self.bbox_metric = DetMetric() if compute_bbox_metric else None
self.main_indicator = main_indicator
self.box_format = box_format
diff --git a/ppstructure/docs/quickstart.md b/ppstructure/docs/quickstart.md
index d206d1d521..7008007599 100644
--- a/ppstructure/docs/quickstart.md
+++ b/ppstructure/docs/quickstart.md
@@ -3,15 +3,17 @@
- [1. 安装依赖包](#1-安装依赖包)
- [2. 便捷使用](#2-便捷使用)
- [2.1 命令行使用](#21-命令行使用)
+ - [2.1.1 图像方向分类+版面分析+表格识别](#211-图像方向分类版面分析表格识别)
- [2.1.1 版面分析+表格识别](#211-版面分析表格识别)
- - [2.1.2 版面分析](#212-版面分析)
- - [2.1.3 表格识别](#213-表格识别)
- - [2.1.4 DocVQA](#214-docvqa)
+ - [2.1.3 版面分析](#213-版面分析)
+ - [2.1.4 表格识别](#214-表格识别)
+ - [2.1.5 DocVQA](#215-docvqa)
- [2.2 代码使用](#22-代码使用)
- - [2.2.1 版面分析+表格识别](#221-版面分析表格识别)
- - [2.2.2 版面分析](#222-版面分析)
- - [2.2.3 表格识别](#223-表格识别)
- - [2.2.4 DocVQA](#224-docvqa)
+ - [2.2.1 图像方向分类版面分析表格识别](#221-图像方向分类版面分析表格识别)
+ - [2.2.2 版面分析+表格识别](#222-版面分析表格识别)
+ - [2.2.3 版面分析](#223-版面分析)
+ - [2.2.4 表格识别](#224-表格识别)
+ - [2.2.5 DocVQA](#225-docvqa)
- [2.3 返回结果说明](#23-返回结果说明)
- [2.3.1 版面分析+表格识别](#231-版面分析表格识别)
- [2.3.2 DocVQA](#232-docvqa)
@@ -36,25 +38,31 @@ pip install paddlenlp
### 2.1 命令行使用
+#### 2.1.1 图像方向分类+版面分析+表格识别
+```bash
+paddleocr --image_dir=PaddleOCR/ppstructure/docs/table/1.png --type=structure --image_orientation=true
+```
+
+
#### 2.1.1 版面分析+表格识别
```bash
paddleocr --image_dir=PaddleOCR/ppstructure/docs/table/1.png --type=structure
```
-
-#### 2.1.2 版面分析
+
+#### 2.1.3 版面分析
```bash
paddleocr --image_dir=PaddleOCR/ppstructure/docs/table/1.png --type=structure --table=false --ocr=false
```
-
-#### 2.1.3 表格识别
+
+#### 2.1.4 表格识别
```bash
paddleocr --image_dir=PaddleOCR/ppstructure/docs/table/table.jpg --type=structure --layout=false
```
-
-#### 2.1.4 DocVQA
+
+#### 2.1.5 DocVQA
请参考:[文档视觉问答](../vqa/README.md)。
@@ -62,7 +70,36 @@ paddleocr --image_dir=PaddleOCR/ppstructure/docs/table/table.jpg --type=structur
### 2.2 代码使用
-#### 2.2.1 版面分析+表格识别
+#### 2.2.1 图像方向分类版面分析表格识别
+
+```python
+import os
+import cv2
+from paddleocr import PPStructure,draw_structure_result,save_structure_res
+
+table_engine = PPStructure(show_log=True, image_orientation=True)
+
+save_folder = './output'
+img_path = 'PaddleOCR/ppstructure/docs/table/1.png'
+img = cv2.imread(img_path)
+result = table_engine(img)
+save_structure_res(result, save_folder,os.path.basename(img_path).split('.')[0])
+
+for line in result:
+ line.pop('img')
+ print(line)
+
+from PIL import Image
+
+font_path = 'PaddleOCR/doc/fonts/simfang.ttf' # PaddleOCR下提供字体包
+image = Image.open(img_path).convert('RGB')
+im_show = draw_structure_result(image, result,font_path=font_path)
+im_show = Image.fromarray(im_show)
+im_show.save('result.jpg')
+```
+
+
+#### 2.2.2 版面分析+表格识别
```python
import os
@@ -90,8 +127,8 @@ im_show = Image.fromarray(im_show)
im_show.save('result.jpg')
```
-
-#### 2.2.2 版面分析
+
+#### 2.2.3 版面分析
```python
import os
@@ -111,8 +148,8 @@ for line in result:
print(line)
```
-
-#### 2.2.3 表格识别
+
+#### 2.2.4 表格识别
```python
import os
@@ -132,8 +169,8 @@ for line in result:
print(line)
```
-
-#### 2.2.4 DocVQA
+
+#### 2.2.5 DocVQA
请参考:[文档视觉问答](../vqa/README.md)。
@@ -154,10 +191,10 @@ PP-Structure的返回结果为一个dict组成的list,示例如下
```
dict 里各个字段说明如下
-| 字段 | 说明 |
-| --------------- |-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
-|type| 图片区域的类型 |
-|bbox| 图片区域的在原图的坐标,分别[左上角x,左上角y,右下角x,右下角y] |
+| 字段 | 说明|
+| --- |---|
+|type| 图片区域的类型 |
+|bbox| 图片区域的在原图的坐标,分别[左上角x,左上角y,右下角x,右下角y]|
|res| 图片区域的OCR或表格识别结果。 表格: 一个dict,字段说明如下 `html`: 表格的HTML字符串 在代码使用模式下,前向传入return_ocr_result_in_table=True可以拿到表格中每个文本的检测识别结果,对应为如下字段: `boxes`: 文本检测坐标 `rec_res`: 文本识别结果。 OCR: 一个包含各个单行文字的检测坐标和识别结果的元组 |
运行完成后,每张图片会在`output`字段指定的目录下有一个同名目录,图片里的每个表格会存储为一个excel,图片区域会被裁剪之后保存下来,excel文件和图片名为表格在图片里的坐标。
@@ -178,20 +215,26 @@ dict 里各个字段说明如下
### 2.4 参数说明
-| 字段 | 说明 | 默认值 |
-|----------------------|----------------------------------------------------------------------------------------------------------------------------------------------------|---------------------------------------------------------|
-| output | excel和识别结果保存的地址 | ./output/table |
-| table_max_len | 表格结构模型预测时,图像的长边resize尺度 | 488 |
-| table_model_dir | 表格结构模型 inference 模型地址 | None |
-| table_char_dict_path | 表格结构模型所用字典地址 | ../ppocr/utils/dict/table_structure_dict.txt |
-| layout_path_model | 版面分析模型模型地址,可以为在线地址或者本地地址,当为本地地址时,需要指定 layout_label_map, 命令行模式下可通过--layout_label_map='{0: "Text", 1: "Title", 2: "List", 3:"Table", 4:"Figure"}' 指定 | lp://PubLayNet/ppyolov2_r50vd_dcn_365e_publaynet/config |
-| layout_label_map | 版面分析模型模型label映射字典 | None |
-| model_name_or_path | VQA SER模型地址 | None |
-| max_seq_length | VQA SER模型最大支持token长度 | 512 |
-| label_map_path | VQA SER 标签文件地址 | ./vqa/labels/labels_ser.txt |
-| layout | 前向中是否执行版面分析 | True |
-| table | 前向中是否执行表格识别 | True |
-| ocr | 对于版面分析中的非表格区域,是否执行ocr。当layout为False时会被自动设置为False | True |
-| structure_version | 表格结构化模型版本,可选 PP-STRUCTURE。PP-STRUCTURE支持表格结构化模型 | PP-STRUCTURE |
+| 字段 | 说明 | 默认值 |
+|---|---|---|
+| output | 结果保存地址 | ./output/table |
+| table_max_len | 表格结构模型预测时,图像的长边resize尺度 | 488 |
+| table_model_dir | 表格结构模型 inference 模型地址| None |
+| table_char_dict_path | 表格结构模型所用字典地址 | ../ppocr/utils/dict/table_structure_dict.txt |
+| merge_no_span_structure | 表格识别模型中,是否对'\'和'\ | ' 进行合并 | False |
+| layout_model_dir | 版面分析模型 inference 模型地址 | None |
+| layout_dict_path | 版面分析模型字典| ../ppocr/utils/dict/layout_publaynet_dict.txt |
+| layout_score_threshold | 版面分析模型检测框阈值| 0.5|
+| layout_nms_threshold | 版面分析模型nms阈值| 0.5|
+| vqa_algorithm | vqa模型算法| LayoutXLM|
+| ser_model_dir | ser模型 inference 模型地址| None|
+| ser_dict_path | ser模型字典| ../train_data/XFUND/class_list_xfun.txt|
+| mode | structure or vqa | structure |
+| image_orientation | 前向中是否执行图像方向分类 | False |
+| layout | 前向中是否执行版面分析 | True |
+| table | 前向中是否执行表格识别 | True |
+| ocr | 对于版面分析中的非表格区域,是否执行ocr。当layout为False时会被自动设置为False| True |
+| recovery | 前向中是否执行版面恢复| False |
+| structure_version | 模型版本,可选 PP-structure和PP-structurev2 | PP-structure |
大部分参数和PaddleOCR whl包保持一致,见 [whl包文档](../../doc/doc_ch/whl.md)
diff --git a/ppstructure/docs/quickstart_en.md b/ppstructure/docs/quickstart_en.md
index 98d8d2fc3f..b4dee3f02d 100644
--- a/ppstructure/docs/quickstart_en.md
+++ b/ppstructure/docs/quickstart_en.md
@@ -3,15 +3,17 @@
- [1. Install package](#1-install-package)
- [2. Use](#2-use)
- [2.1 Use by command line](#21-use-by-command-line)
- - [2.1.1 layout analysis + table recognition](#211-layout-analysis--table-recognition)
- - [2.1.2 layout analysis](#212-layout-analysis)
- - [2.1.3 table recognition](#213-table-recognition)
- - [2.1.4 DocVQA](#214-docvqa)
+ - [2.1.1 image orientation + layout analysis + table recognition](#211-image-orientation--layout-analysis--table-recognition)
+ - [2.1.2 layout analysis + table recognition](#212-layout-analysis--table-recognition)
+ - [2.1.3 layout analysis](#213-layout-analysis)
+ - [2.1.4 table recognition](#214-table-recognition)
+ - [2.1.5 DocVQA](#215-docvqa)
- [2.2 Use by code](#22-use-by-code)
- - [2.2.1 layout analysis + table recognition](#221-layout-analysis--table-recognition)
- - [2.2.2 layout analysis](#222-layout-analysis)
- - [2.2.3 table recognition](#223-table-recognition)
- - [2.2.4 DocVQA](#224-docvqa)
+ - [2.2.1 image orientation + layout analysis + table recognition](#221-image-orientation--layout-analysis--table-recognition)
+ - [2.2.2 layout analysis + table recognition](#222-layout-analysis--table-recognition)
+ - [2.2.3 layout analysis](#223-layout-analysis)
+ - [2.2.4 table recognition](#224-table-recognition)
+ - [2.2.5 DocVQA](#225-docvqa)
- [2.3 Result description](#23-result-description)
- [2.3.1 layout analysis + table recognition](#231-layout-analysis--table-recognition)
- [2.3.2 DocVQA](#232-docvqa)
@@ -36,25 +38,31 @@ pip install paddlenlp
### 2.1 Use by command line
-#### 2.1.1 layout analysis + table recognition
+#### 2.1.1 image orientation + layout analysis + table recognition
+```bash
+paddleocr --image_dir=PaddleOCR/ppstructure/docs/table/1.png --type=structure --image_orientation=true
+```
+
+
+#### 2.1.2 layout analysis + table recognition
```bash
paddleocr --image_dir=PaddleOCR/ppstructure/docs/table/1.png --type=structure
```
-
-#### 2.1.2 layout analysis
+
+#### 2.1.3 layout analysis
```bash
paddleocr --image_dir=PaddleOCR/ppstructure/docs/table/1.png --type=structure --table=false --ocr=false
```
-
-#### 2.1.3 table recognition
+
+#### 2.1.4 table recognition
```bash
paddleocr --image_dir=PaddleOCR/ppstructure/docs/table/table.jpg --type=structure --layout=false
```
-
-#### 2.1.4 DocVQA
+
+#### 2.1.5 DocVQA
Please refer to: [Documentation Visual Q&A](../vqa/README.md) .
@@ -62,7 +70,36 @@ Please refer to: [Documentation Visual Q&A](../vqa/README.md) .
### 2.2 Use by code
-#### 2.2.1 layout analysis + table recognition
+#### 2.2.1 image orientation + layout analysis + table recognition
+
+```python
+import os
+import cv2
+from paddleocr import PPStructure,draw_structure_result,save_structure_res
+
+table_engine = PPStructure(show_log=True, image_orientation=True)
+
+save_folder = './output'
+img_path = 'PaddleOCR/ppstructure/docs/table/1.png'
+img = cv2.imread(img_path)
+result = table_engine(img)
+save_structure_res(result, save_folder,os.path.basename(img_path).split('.')[0])
+
+for line in result:
+ line.pop('img')
+ print(line)
+
+from PIL import Image
+
+font_path = 'PaddleOCR/doc/fonts/simfang.ttf' # PaddleOCR下提供字体包
+image = Image.open(img_path).convert('RGB')
+im_show = draw_structure_result(image, result,font_path=font_path)
+im_show = Image.fromarray(im_show)
+im_show.save('result.jpg')
+```
+
+
+#### 2.2.2 layout analysis + table recognition
```python
import os
@@ -90,8 +127,8 @@ im_show = Image.fromarray(im_show)
im_show.save('result.jpg')
```
-
-#### 2.2.2 layout analysis
+
+#### 2.2.3 layout analysis
```python
import os
@@ -111,8 +148,8 @@ for line in result:
print(line)
```
-
-#### 2.2.3 table recognition
+
+#### 2.2.4 table recognition
```python
import os
@@ -132,8 +169,8 @@ for line in result:
print(line)
```
-
-#### 2.2.4 DocVQA
+
+#### 2.2.5 DocVQA
Please refer to: [Documentation Visual Q&A](../vqa/README.md) .
@@ -155,8 +192,8 @@ The return of PP-Structure is a list of dicts, the example is as follows:
```
Each field in dict is described as follows:
-| field | description |
-| --------------- |--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
+| field | description |
+| --- |---|
|type| Type of image area. |
|bbox| The coordinates of the image area in the original image, respectively [upper left corner x, upper left corner y, lower right corner x, lower right corner y]. |
|res| OCR or table recognition result of the image area. table: a dict with field descriptions as follows: `html`: html str of table. In the code usage mode, set return_ocr_result_in_table=True whrn call can get the detection and recognition results of each text in the table area, corresponding to the following fields: `boxes`: text detection boxes. `rec_res`: text recognition results. OCR: A tuple containing the detection boxes and recognition results of each single text. |
@@ -178,19 +215,26 @@ Please refer to: [Documentation Visual Q&A](../vqa/README.md) .
### 2.4 Parameter Description
-| field | description | default |
-|----------------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|---------------------------------------------------------|
-| output | The save path of result | ./output/table |
-| table_max_len | When the table structure model predicts, the long side of the image | 488 |
-| table_model_dir | the path of table structure model | None |
-| table_char_dict_path | the dict path of table structure model | ../ppocr/utils/dict/table_structure_dict.txt |
-| layout_path_model | The model path of the layout analysis model, which can be an online address or a local path. When it is a local path, layout_label_map needs to be set. In command line mode, use --layout_label_map='{0: "Text", 1: "Title", 2: "List", 3:"Table", 4:"Figure"}' | lp://PubLayNet/ppyolov2_r50vd_dcn_365e_publaynet/config |
-| layout_label_map | Layout analysis model model label mapping dictionary path | None |
-| model_name_or_path | the model path of VQA SER model | None |
-| max_seq_length | the max token length of VQA SER model | 512 |
-| label_map_path | the label path of VQA SER model | ./vqa/labels/labels_ser.txt |
-| layout | Whether to perform layout analysis in forward | True |
-| table | Whether to perform table recognition in forward | True |
-| ocr | Whether to perform ocr for non-table areas in layout analysis. When layout is False, it will be automatically set to False | True |
-| structure_version | table structure Model version number, the current model support list is as follows: PP-STRUCTURE support english table structure model | PP-STRUCTURE |
+| field | description | default |
+|---|---|---|
+| output | result save path | ./output/table |
+| table_max_len | long side of the image resize in table structure model | 488 |
+| table_model_dir | Table structure model inference model path| None |
+| table_char_dict_path | The dictionary path of table structure model | ../ppocr/utils/dict/table_structure_dict.txt |
+| merge_no_span_structure | In the table recognition model, whether to merge '\' and '\ | ' | False |
+| layout_model_dir | Layout analysis model inference model path| None |
+| layout_dict_path | The dictionary path of layout analysis model| ../ppocr/utils/dict/layout_publaynet_dict.txt |
+| layout_score_threshold | The box threshold path of layout analysis model| 0.5|
+| layout_nms_threshold | The nms threshold path of layout analysis model| 0.5|
+| vqa_algorithm | vqa model algorithm| LayoutXLM|
+| ser_model_dir | Ser model inference model path| None|
+| ser_dict_path | The dictionary path of Ser model| ../train_data/XFUND/class_list_xfun.txt|
+| mode | structure or vqa | structure |
+| image_orientation | Whether to perform image orientation classification in forward | False |
+| layout | Whether to perform layout analysis in forward | True |
+| table | Whether to perform table recognition in forward | True |
+| ocr | Whether to perform ocr for non-table areas in layout analysis. When layout is False, it will be automatically set to False| True |
+| recovery | Whether to perform layout recovery in forward| False |
+| structure_version | Structure version, optional PP-structure and PP-structurev2 | PP-structure |
+
Most of the parameters are consistent with the PaddleOCR whl package, see [whl package documentation](../../doc/doc_en/whl.md)
diff --git a/ppstructure/predict_system.py b/ppstructure/predict_system.py
index 608f4d2fb3..053a8aac00 100644
--- a/ppstructure/predict_system.py
+++ b/ppstructure/predict_system.py
@@ -27,7 +27,6 @@ import numpy as np
import time
import logging
from copy import deepcopy
-from attrdict import AttrDict
from ppocr.utils.utility import get_image_file_list, check_and_read_gif
from ppocr.utils.logging import get_logger
@@ -44,6 +43,13 @@ class StructureSystem(object):
def __init__(self, args):
self.mode = args.mode
self.recovery = args.recovery
+
+ self.image_orientation_predictor = None
+ if args.image_orientation:
+ import paddleclas
+ self.image_orientation_predictor = paddleclas.PaddleClas(
+ model_name="text_image_orientation")
+
if self.mode == 'structure':
if not args.show_log:
logger.setLevel(logging.INFO)
@@ -74,6 +80,7 @@ class StructureSystem(object):
def __call__(self, img, return_ocr_result_in_table=False):
time_dict = {
+ 'image_orientation': 0,
'layout': 0,
'table': 0,
'table_match': 0,
@@ -83,6 +90,20 @@ class StructureSystem(object):
'all': 0
}
start = time.time()
+ if self.image_orientation_predictor is not None:
+ tic = time.time()
+ cls_result = self.image_orientation_predictor.predict(
+ input_data=img)
+ cls_res = next(cls_result)
+ angle = cls_res[0]['label_names'][0]
+ cv_rotate_code = {
+ '90': cv2.ROTATE_90_COUNTERCLOCKWISE,
+ '180': cv2.ROTATE_180,
+ '270': cv2.ROTATE_90_CLOCKWISE
+ }
+ img = cv2.rotate(img, cv_rotate_code[angle])
+ toc = time.time()
+ time_dict['image_orientation'] = toc - tic
if self.mode == 'structure':
ori_im = img.copy()
if self.layout_predictor is not None:
@@ -121,7 +142,10 @@ class StructureSystem(object):
roi_img)
time_dict['det'] += ocr_time_dict['det']
time_dict['rec'] += ocr_time_dict['rec']
- # remove style char
+
+ # remove style char,
+ # when using the recognition model trained on the PubtabNet dataset,
+ # it will recognize the text format in the table, such as
style_token = [
'', '', '', '', '',
'', '', '', '',
@@ -198,7 +222,6 @@ def main(args):
if img is None:
logger.error("error in loading image:{}".format(image_file))
continue
- starttime = time.time()
res, time_dict = structure_sys(img)
if structure_sys.mode == 'structure':
@@ -213,8 +236,7 @@ def main(args):
logger.info('result save to {}'.format(img_save_path))
if args.recovery:
convert_info_docx(img, res, save_folder, img_name)
- elapse = time.time() - starttime
- logger.info("Predict time : {:.3f}s".format(elapse))
+ logger.info("Predict time : {:.3f}s".format(time_dict['all']))
if __name__ == "__main__":
diff --git a/ppstructure/utility.py b/ppstructure/utility.py
index 597d978516..fcba52b27d 100644
--- a/ppstructure/utility.py
+++ b/ppstructure/utility.py
@@ -62,6 +62,11 @@ def init_args():
type=str,
default='structure',
help='structure and vqa is supported')
+ parser.add_argument(
+ "--image_orientation",
+ type=bool,
+ default=False,
+ help='Whether to enable image orientation recognition')
parser.add_argument(
"--layout",
type=str2bool,
From ce321153e76096a914d00ac0967c83daaf91bcd3 Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Sun, 14 Aug 2022 09:01:49 +0000
Subject: [PATCH 18/49] rm unused code
---
deploy/cpp_infer/src/postprocess_op.cpp | 4 +-
ppocr/utils/visual.py | 1 +
ppstructure/docs/quickstart.md | 4 +-
ppstructure/table/eval_table.py | 3 +-
ppstructure/table/predict_table.py | 9 ++-
ppstructure/table/table_master_match.py | 74 -------------------------
tools/train.py | 1 +
7 files changed, 14 insertions(+), 82 deletions(-)
diff --git a/deploy/cpp_infer/src/postprocess_op.cpp b/deploy/cpp_infer/src/postprocess_op.cpp
index 8d7af6474c..551f98a166 100644
--- a/deploy/cpp_infer/src/postprocess_op.cpp
+++ b/deploy/cpp_infer/src/postprocess_op.cpp
@@ -400,7 +400,7 @@ void TablePostProcessor::Run(
score += char_score;
rec_html_tags.push_back(html_tag);
// box
- if (html_tag == "| " || html_tag == " | " || html_tag == " | | ") {
for (int point_idx = 0; point_idx < loc_preds_shape[2];
point_idx += 2) {
std::vector point(2, 0);
@@ -416,7 +416,7 @@ void TablePostProcessor::Run(
}
}
score /= count;
- if (isnan(score) || rec_boxes.size() == 0 || rec_html_tags.size() == 0) {
+ if (isnan(score) || rec_boxes.size() == 0) {
score = -1;
}
rec_scores.push_back(score);
diff --git a/ppocr/utils/visual.py b/ppocr/utils/visual.py
index 030d1c38d2..5bd805ea6e 100644
--- a/ppocr/utils/visual.py
+++ b/ppocr/utils/visual.py
@@ -114,6 +114,7 @@ def draw_re_results(image,
def draw_rectangle(img_path, boxes):
+ boxes = np.array(boxes)
img = cv2.imread(img_path)
img_show = img.copy()
for box in boxes.astype(int):
diff --git a/ppstructure/docs/quickstart.md b/ppstructure/docs/quickstart.md
index 7008007599..f4645bdfe0 100644
--- a/ppstructure/docs/quickstart.md
+++ b/ppstructure/docs/quickstart.md
@@ -4,7 +4,7 @@
- [2. 便捷使用](#2-便捷使用)
- [2.1 命令行使用](#21-命令行使用)
- [2.1.1 图像方向分类+版面分析+表格识别](#211-图像方向分类版面分析表格识别)
- - [2.1.1 版面分析+表格识别](#211-版面分析表格识别)
+ - [2.1.2 版面分析+表格识别](#212-版面分析表格识别)
- [2.1.3 版面分析](#213-版面分析)
- [2.1.4 表格识别](#214-表格识别)
- [2.1.5 DocVQA](#215-docvqa)
@@ -44,7 +44,7 @@ paddleocr --image_dir=PaddleOCR/ppstructure/docs/table/1.png --type=structure --
```
-#### 2.1.1 版面分析+表格识别
+#### 2.1.2 版面分析+表格识别
```bash
paddleocr --image_dir=PaddleOCR/ppstructure/docs/table/1.png --type=structure
```
diff --git a/ppstructure/table/eval_table.py b/ppstructure/table/eval_table.py
index 435d693223..4fc16b5d4c 100755
--- a/ppstructure/table/eval_table.py
+++ b/ppstructure/table/eval_table.py
@@ -1,4 +1,4 @@
-# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved.
+# Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
@@ -11,6 +11,7 @@
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
+
import os
import sys
diff --git a/ppstructure/table/predict_table.py b/ppstructure/table/predict_table.py
index f580213753..e94347d861 100644
--- a/ppstructure/table/predict_table.py
+++ b/ppstructure/table/predict_table.py
@@ -117,7 +117,6 @@ class TableSystem(object):
pred_html = self.match(structure_res, dt_boxes, rec_res)
toc = time.time()
time_dict['match'] = toc - tic
- # pred_html = self.match(1, 1, 1,img_name)
result['html'] = pred_html
if self.benchmark:
self.autolog.times.end(stamp=True)
@@ -212,8 +211,12 @@ def main(args):
elapse = time.time() - starttime
logger.info("Predict time : {:.3f}s".format(elapse))
- # img = predict_strture.draw_rectangle(image_file, pred_res['cell_bbox'], use_xywh)
- img = utility.draw_boxes(cv2.imread(image_file), pred_res['cell_bbox'])
+ if len(pred_res['cell_bbox']) > 0 and len(pred_res['cell_bbox'][
+ 0]) == 4:
+ img = predict_strture.draw_rectangle(image_file,
+ pred_res['cell_bbox'])
+ else:
+ img = utility.draw_boxes(img, pred_res['cell_bbox'])
img_save_path = os.path.join(args.output, os.path.basename(image_file))
cv2.imwrite(img_save_path, img)
diff --git a/ppstructure/table/table_master_match.py b/ppstructure/table/table_master_match.py
index 6a4c4e9dc6..163b13063b 100644
--- a/ppstructure/table/table_master_match.py
+++ b/ppstructure/table/table_master_match.py
@@ -273,10 +273,6 @@ def sort_bbox(end2end_xywh_bboxes, no_match_end2end_indexes):
end2end_sorted_idx_list, end2end_sorted_bbox_list \
= flatten(sorted_groups, sorted_bbox_groups)
- # check sorted
- #img = cv2.imread('/data_0/yejiaquan/data/TableRecognization/singleVal/PMC3286376_004_00.png')
- #img = drawBboxAfterSorted(img, sorted_groups, sorted_bbox_groups)
-
return end2end_sorted_idx_list, end2end_sorted_bbox_list, sorted_groups, sorted_bbox_groups
@@ -302,9 +298,6 @@ def get_bboxes_list(end2end_result, structure_master_result):
# structure master
src_bboxes = structure_master_result['bbox']
src_bboxes = remove_empty_bboxes(src_bboxes)
- # structure_master_xywh_bboxes = src_bboxes
- # xyxy_bboxes = xywh2xyxy(src_bboxes)
- # structure_master_xyxy_bboxes = xyxy_bboxes
structure_master_xyxy_bboxes = src_bboxes
xywh_bbox = xyxy2xywh(src_bboxes)
structure_master_xywh_bboxes = xywh_bbox
@@ -410,64 +403,6 @@ def extra_match(no_match_end2end_indexes, master_bbox_nums):
return extra_match_list
-def match_visual(file_name,
- match_list,
- end2end_xyxy,
- master_xyxy,
- prex='ordinary_match'):
- """
- Show the match result by xyxy coord style.
- :param file_name:
- :param match_list:
- :param end2end_xyxy:
- :param master_xyxy:
- :param prex:
- :return:
- """
- folder = ''
- save_folder = '/data_0/cache'
- file_path = os.path.join(folder, file_name)
- img_end2end = cv2.imread(file_path)
- img_master = copy.deepcopy(img_end2end)
- text_color = (0, 0, 255)
- bbox_color = (255, 0, 0)
- master_nums = len(master_xyxy)
-
- for idx, match_group in enumerate(match_list):
- end2end_idx, master_index = match_group[0], match_group[1]
-
- # master_index larger than master_nums, did not draw master bbox.
- if master_index < master_nums:
- # draw master
- master_bbox = master_xyxy[master_index]
- img_master = cv2.rectangle(
- img_master, (int(master_bbox[0]), int(master_bbox[1])),
- (int(master_bbox[2]), int(master_bbox[3])),
- bbox_color,
- thickness=1)
- master_text_coord = (int(master_bbox[0]) - 4, int(master_bbox[1]))
- img_master = cv2.putText(img_master,
- str(master_index), master_text_coord, 1, 1,
- text_color, 2)
-
- # draw end2end
- end2end_bbox = end2end_xyxy[end2end_idx]
- img_end2end = cv2.rectangle(
- img_end2end, (int(end2end_bbox[0]), int(end2end_bbox[1])),
- (int(end2end_bbox[2]), int(end2end_bbox[3])),
- bbox_color,
- thickness=1)
- end2end_text_coord = (int(end2end_bbox[0]) - 4, int(end2end_bbox[1]))
- # write end2end bbox matching master bbox's index
- img_end2end = cv2.putText(img_end2end,
- str(master_index), end2end_text_coord, 1, 1,
- text_color, 2)
-
- img = np.hstack([img_end2end, img_master])
- save_path = os.path.join(save_folder, '{}_matchShow.png'.format(prex))
- cv2.imwrite(save_path, img)
-
-
def get_match_dict(match_list):
"""
Convert match_list to a dict, where key is master bbox's index, value is end2end bbox index.
@@ -555,8 +490,6 @@ def merge_span_token(master_token_list):
pattern
' | ' + ' | '
"""
- # tmp = master_token_list[pointer] + master_token_list[pointer+1] + master_token_list[pointer+2] + \
- # master_token_list[pointer+3]
tmp = ''.join(master_token_list[pointer:pointer + 3 + 1])
pointer += 4
new_master_token_list.append(tmp)
@@ -569,8 +502,6 @@ def merge_span_token(master_token_list):
pattern
' | ' + ' | '
"""
- # tmp = master_token_list[pointer] + master_token_list[pointer+1] + \
- # master_token_list[pointer+2] + master_token_list[pointer+3] + master_token_list[pointer+4]
tmp = ''.join(master_token_list[pointer:pointer + 4 + 1])
pointer += 5
new_master_token_list.append(tmp)
@@ -909,11 +840,6 @@ class Matcher:
'sorted_bboxes_groups': sorted_bboxes_groups
}
- # ordinary match show
- # match_visual(file_name, match_list, end2end_xyxy_bboxes, structure_master_xyxy_bboxes, prex='ordinary_match')
- # extra match show
- # match_visual(file_name, match_list_add_extra_match, end2end_xyxy_bboxes, structure_master_xyxy_bboxes, prex='extra_match')
-
# format output
match_result_dict = self._format(match_result_dict, file_name)
diff --git a/tools/train.py b/tools/train.py
index 0c881ecae8..a46cd67cb9 100755
--- a/tools/train.py
+++ b/tools/train.py
@@ -125,6 +125,7 @@ def main(config, device, logger, vdl_writer):
logger.info('convert_sync_batchnorm')
model = apply_to_static(model, config, logger)
+ logger.info(model)
# build loss
loss_class = build_loss(config['Loss'])
From 444080a0cddfa783b5b6e597d4f2b5b0bfb97e1e Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Mon, 15 Aug 2022 08:51:07 +0000
Subject: [PATCH 19/49] update vis
---
tools/infer_table.py | 7 ++++++-
1 file changed, 6 insertions(+), 1 deletion(-)
diff --git a/tools/infer_table.py b/tools/infer_table.py
index 70dc6205d3..6dde5d67d0 100644
--- a/tools/infer_table.py
+++ b/tools/infer_table.py
@@ -37,6 +37,7 @@ from ppocr.postprocess import build_post_process
from ppocr.utils.save_load import load_model
from ppocr.utils.utility import get_image_file_list
from ppocr.utils.visual import draw_rectangle
+from tools.infer.utility import draw_boxes
import tools.program as program
import cv2
@@ -105,9 +106,13 @@ def main(config, device, logger, vdl_writer):
f_w.write("result: {}, {}\n".format(structure_str_list,
bbox_list_str))
- img = draw_rectangle(file, bbox_list)
+ if len(bbox_list) > 0 and len(bbox_list[0]) == 4:
+ img = draw_rectangle(file, bbox_list)
+ else:
+ img = draw_boxes(cv2.imread(file), bbox_list)
cv2.imwrite(
os.path.join(save_res_path, os.path.basename(file)), img)
+ logger.info('save result to {}'.format(save_res_path))
logger.info("success!")
From ebe3e885c0e098305149adb86579b718f050845a Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Mon, 15 Aug 2022 09:00:07 +0000
Subject: [PATCH 20/49] Support variable length input
---
ppocr/modeling/necks/csp_pan.py | 3 +--
1 file changed, 1 insertion(+), 2 deletions(-)
diff --git a/ppocr/modeling/necks/csp_pan.py b/ppocr/modeling/necks/csp_pan.py
index 625508e995..f4f8547f7d 100755
--- a/ppocr/modeling/necks/csp_pan.py
+++ b/ppocr/modeling/necks/csp_pan.py
@@ -304,9 +304,8 @@ class CSPPAN(nn.Layer):
for idx in range(len(self.in_channels) - 1, 0, -1):
feat_heigh = inner_outs[0]
feat_low = inputs[idx - 1]
-
upsample_feat = F.upsample(
- feat_heigh, size=feat_low.shape[2:4], mode="nearest")
+ feat_heigh, size=paddle.shape(feat_low)[2:4], mode="nearest")
inner_out = self.top_down_blocks[len(self.in_channels) - 1 - idx](
paddle.concat([upsample_feat, feat_low], 1))
From 7efee821bcd5feda5ca1aade47ed954c3252cf76 Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Mon, 15 Aug 2022 09:00:35 +0000
Subject: [PATCH 21/49] Support variable length input
---
tools/export_model.py | 2 ++
1 file changed, 2 insertions(+)
diff --git a/tools/export_model.py b/tools/export_model.py
index 2443d66ca2..3d30fa77ea 100755
--- a/tools/export_model.py
+++ b/tools/export_model.py
@@ -140,6 +140,8 @@ def export_single_model(model,
infer_shape = [3, 488, 488]
if arch_config["algorithm"] == "TableMaster":
infer_shape = [3, 480, 480]
+ if arch_config["algorithm"] == "SLANet":
+ infer_shape = [3, -1, -1]
model = to_static(
model,
input_spec=[
From a8efe28f77a72d780eeb05cb556388e39f170d0e Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Mon, 15 Aug 2022 09:26:01 +0000
Subject: [PATCH 22/49] add html to tablemaster result
---
ppstructure/table/table_master_match.py | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/ppstructure/table/table_master_match.py b/ppstructure/table/table_master_match.py
index 163b13063b..7a7208d4a9 100644
--- a/ppstructure/table/table_master_match.py
+++ b/ppstructure/table/table_master_match.py
@@ -949,5 +949,5 @@ class TableMasterMatcher(Matcher):
match_results = self.match()
merged_results = self.get_merge_result(match_results)
pred_html = merged_results[img_name]
- # pred_html = ''
+ pred_html = ''
return pred_html
From 4369552ea2e71ba2a5e0277134ea7bdfef867fd3 Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Mon, 15 Aug 2022 10:04:06 +0000
Subject: [PATCH 23/49] add in and out box_format
---
configs/table/SLANet.yml | 6 ++++--
configs/table/table_master.yml | 6 ++++--
ppocr/data/imaug/label_ops.py | 29 ++++++++++++++++++-----------
3 files changed, 26 insertions(+), 15 deletions(-)
diff --git a/configs/table/SLANet.yml b/configs/table/SLANet.yml
index 4858c71c28..2264eb14d2 100644
--- a/configs/table/SLANet.yml
+++ b/configs/table/SLANet.yml
@@ -86,7 +86,8 @@ Train:
loc_reg_num: *loc_reg_num
max_text_length: *max_text_length
- TableBoxEncode:
- box_format: *box_format
+ in_box_format: *box_format
+ out_box_format: *box_format
- ResizeTableImage:
max_len: 488
- NormalizeImage:
@@ -121,7 +122,8 @@ Eval:
loc_reg_num: *loc_reg_num
max_text_length: *max_text_length
- TableBoxEncode:
- box_format: *box_format
+ in_box_format: *box_format
+ out_box_format: *box_format
- ResizeTableImage:
max_len: 488
- NormalizeImage:
diff --git a/configs/table/table_master.yml b/configs/table/table_master.yml
index 1f50d20bfa..df437f7c95 100755
--- a/configs/table/table_master.yml
+++ b/configs/table/table_master.yml
@@ -90,7 +90,8 @@ Train:
- PaddingTableImage:
size: [480, 480]
- TableBoxEncode:
- box_format: *box_format
+ in_box_format: *box_format
+ out_box_format: *box_format
- NormalizeImage:
scale: 1./255.
mean: [0.5, 0.5, 0.5]
@@ -126,7 +127,8 @@ Eval:
- PaddingTableImage:
size: [480, 480]
- TableBoxEncode:
- box_format: *box_format
+ in_box_format: *box_format
+ out_box_format: *box_format
- NormalizeImage:
scale: 1./255.
mean: [0.5, 0.5, 0.5]
diff --git a/ppocr/data/imaug/label_ops.py b/ppocr/data/imaug/label_ops.py
index d98facb5cb..73d4ca430f 100644
--- a/ppocr/data/imaug/label_ops.py
+++ b/ppocr/data/imaug/label_ops.py
@@ -749,28 +749,35 @@ class TableMasterLabelEncode(TableLabelEncode):
class TableBoxEncode(object):
- def __init__(self, box_format='xyxy', **kwargs):
+ def __init__(self, in_box_format='xyxy', out_box_format='xyxy', **kwargs):
assert box_format in ['xywh', 'xyxy', 'xyxyxyxy']
- self.box_format = box_format
+ self.in_box_format = in_box_format
+ self.out_box_format = out_box_format
def __call__(self, data):
img_height, img_width = data['image'].shape[:2]
bboxes = data['bboxes']
- if self.box_format == 'xywh' and bboxes.shape[1] == 4:
- bboxes = self.xyxy2xywh(bboxes)
+ if self.in_box_format != self.out_box_format:
+ if self.out_box_format == 'xywh':
+ if self.in_box_format == 'xyxyxyxy':
+ bboxes = self.xyxyxyxy2xywh(bboxes)
+ elif self.in_box_format == 'xyxy':
+ bboxes = self.xyxy2xywh(bboxes)
+
bboxes[:, 0::2] /= img_width
bboxes[:, 1::2] /= img_height
data['bboxes'] = bboxes
return data
+ def xyxyxyxy2xywh(self, boxes):
+ new_bboxes = np.zeros([len(bboxes), 4])
+ new_bboxes[:, 0] = bboxes[:, 0::2].min() # x1
+ new_bboxes[:, 1] = bboxes[:, 1::2].min() # y1
+ new_bboxes[:, 2] = bboxes[:, 0::2].max() - new_bboxes[:, 0] # w
+ new_bboxes[:, 3] = bboxes[:, 1::2].max() - new_bboxes[:, 1] # h
+ return new_bboxes
+
def xyxy2xywh(self, bboxes):
- """
- Convert coord (x1,y1,x2,y2) to (x,y,w,h).
- where (x1,y1) is top-left, (x2,y2) is bottom-right.
- (x,y) is bbox center and (w,h) is width and height.
- :param bboxes: (x1, y1, x2, y2)
- :return:
- """
new_bboxes = np.empty_like(bboxes)
new_bboxes[:, 0] = (bboxes[:, 0] + bboxes[:, 2]) / 2 # x center
new_bboxes[:, 1] = (bboxes[:, 1] + bboxes[:, 3]) / 2 # y center
From 6460985d8d0068da6402eb59aa2cee1d597b3612 Mon Sep 17 00:00:00 2001
From: littletomatodonkey
Date: Tue, 16 Aug 2022 10:47:31 +0800
Subject: [PATCH 24/49] add vi-layoutxlm (#7209)
---
ppstructure/vqa/predict_vqa_token_ser.py | 6 +-
.../vi_layoutxlm_ser/train_infer_python.txt | 59 +++++++++++++++++++
test_tipc/prepare.sh | 4 +-
3 files changed, 66 insertions(+), 3 deletions(-)
create mode 100644 test_tipc/configs/vi_layoutxlm_ser/train_infer_python.txt
diff --git a/ppstructure/vqa/predict_vqa_token_ser.py b/ppstructure/vqa/predict_vqa_token_ser.py
index 855be42de3..7647af9d10 100644
--- a/ppstructure/vqa/predict_vqa_token_ser.py
+++ b/ppstructure/vqa/predict_vqa_token_ser.py
@@ -41,7 +41,11 @@ logger = get_logger()
class SerPredictor(object):
def __init__(self, args):
self.ocr_engine = PaddleOCR(
- use_angle_cls=False, show_log=False, use_gpu=args.use_gpu)
+ use_angle_cls=args.use_angle_cls,
+ det_model_dir=args.det_model_dir,
+ rec_model_dir=args.rec_model_dir,
+ show_log=False,
+ use_gpu=args.use_gpu)
pre_process_list = [{
'VQATokenLabelEncode': {
diff --git a/test_tipc/configs/vi_layoutxlm_ser/train_infer_python.txt b/test_tipc/configs/vi_layoutxlm_ser/train_infer_python.txt
new file mode 100644
index 0000000000..59d3474611
--- /dev/null
+++ b/test_tipc/configs/vi_layoutxlm_ser/train_infer_python.txt
@@ -0,0 +1,59 @@
+===========================train_params===========================
+model_name:vi_layoutxlm_ser
+python:python3.7
+gpu_list:0|0,1
+Global.use_gpu:True|True
+Global.auto_cast:fp32
+Global.epoch_num:lite_train_lite_infer=1|whole_train_whole_infer=17
+Global.save_model_dir:./output/
+Train.loader.batch_size_per_card:lite_train_lite_infer=4|whole_train_whole_infer=8
+Architecture.Backbone.checkpoints:null
+train_model_name:latest
+train_infer_img_dir:ppstructure/docs/vqa/input/zh_val_42.jpg
+null:null
+##
+trainer:norm_train
+norm_train:tools/train.py -c ./configs/kie/vi_layoutxlm/ser_vi_layoutxlm_xfund_zh.yml -o Global.print_batch_step=1 Global.eval_batch_step=[1000,1000] Train.loader.shuffle=false
+pact_train:null
+fpgm_train:null
+distill_train:null
+null:null
+null:null
+##
+===========================eval_params===========================
+eval:null
+null:null
+##
+===========================infer_params===========================
+Global.save_inference_dir:./output/
+Architecture.Backbone.checkpoints:
+norm_export:tools/export_model.py -c ./configs/kie/vi_layoutxlm/ser_vi_layoutxlm_xfund_zh.yml -o
+quant_export:
+fpgm_export:
+distill_export:null
+export1:null
+export2:null
+##
+infer_model:null
+infer_export:null
+infer_quant:False
+inference:ppstructure/vqa/predict_vqa_token_ser.py --vqa_algorithm=LayoutXLM --ser_dict_path=train_data/XFUND/class_list_xfun.txt --output=output --ocr_order_method=tb-yx
+--use_gpu:True|False
+--enable_mkldnn:False
+--cpu_threads:6
+--rec_batch_num:1
+--use_tensorrt:False
+--precision:fp32
+--ser_model_dir:
+--image_dir:./ppstructure/docs/vqa/input/zh_val_42.jpg
+null:null
+--benchmark:False
+null:null
+===========================infer_benchmark_params==========================
+random_infer_input:[{float32,[3,224,224]}]
+===========================train_benchmark_params==========================
+batch_size:4
+fp_items:fp32|fp16
+epoch:3
+--profiler_options:batch_range=[10,20];state=GPU;tracer_option=Default;profile_path=model.profile
+flags:FLAGS_eager_delete_tensor_gb=0.0;FLAGS_fraction_of_gpu_memory_to_use=0.98
diff --git a/test_tipc/prepare.sh b/test_tipc/prepare.sh
index 76543f39e4..259a1159cb 100644
--- a/test_tipc/prepare.sh
+++ b/test_tipc/prepare.sh
@@ -106,7 +106,7 @@ if [ ${MODE} = "benchmark_train" ];then
ln -s ./icdar2015_benckmark ./icdar2015
cd ../
fi
- if [ ${model_name} == "layoutxlm_ser" ]; then
+ if [ ${model_name} == "layoutxlm_ser" ] || [ ${model_name} == "vi_layoutxlm_ser" ]; then
pip install -r ppstructure/vqa/requirements.txt
pip install paddlenlp\>=2.3.5 --force-reinstall -i https://mirrors.aliyun.com/pypi/simple/
wget -nc -P ./train_data/ https://paddleocr.bj.bcebos.com/ppstructure/dataset/XFUND.tar --no-check-certificate
@@ -220,7 +220,7 @@ if [ ${MODE} = "lite_train_lite_infer" ];then
wget -nc -P ./pretrain_models/ https://paddleocr.bj.bcebos.com/rec_r32_gaspin_bilstm_att_train.tar --no-check-certificate
cd ./pretrain_models/ && tar xf rec_r32_gaspin_bilstm_att_train.tar && cd ../
fi
- if [ ${model_name} == "layoutxlm_ser" ]; then
+ if [ ${model_name} == "layoutxlm_ser" ] || [ ${model_name} == "vi_layoutxlm_ser" ]; then
pip install -r ppstructure/vqa/requirements.txt
pip install paddlenlp\>=2.3.5 --force-reinstall -i https://mirrors.aliyun.com/pypi/simple/
wget -nc -P ./train_data/ https://paddleocr.bj.bcebos.com/ppstructure/dataset/XFUND.tar --no-check-certificate
From c1e6558d9e4c9a67221799c940be10e1b1609a83 Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Tue, 16 Aug 2022 03:29:17 +0000
Subject: [PATCH 25/49] update table cpp infer doc
---
deploy/cpp_infer/readme.md | 5 -----
deploy/cpp_infer/readme_ch.md | 5 -----
2 files changed, 10 deletions(-)
diff --git a/deploy/cpp_infer/readme.md b/deploy/cpp_infer/readme.md
index 545924c5ce..2afdf79521 100644
--- a/deploy/cpp_infer/readme.md
+++ b/deploy/cpp_infer/readme.md
@@ -283,13 +283,8 @@ Specifically,
```shell
./build/ppocr --det_model_dir=inference/det_db \
--rec_model_dir=inference/rec_rcnn \
- --cls_model_dir=inference/cls \
--table_model_dir=inference/table \
--image_dir=../../ppstructure/docs/table/table.jpg \
- --use_angle_cls=true \
- --det=true \
- --rec=true \
- --cls=true \
--type=structure \
--table=true
```
diff --git a/deploy/cpp_infer/readme_ch.md b/deploy/cpp_infer/readme_ch.md
index fb994a5b41..d94c95c8c5 100644
--- a/deploy/cpp_infer/readme_ch.md
+++ b/deploy/cpp_infer/readme_ch.md
@@ -292,13 +292,8 @@ CUDNN_LIB_DIR=/your_cudnn_lib_dir
```shell
./build/ppocr --det_model_dir=inference/det_db \
--rec_model_dir=inference/rec_rcnn \
- --cls_model_dir=inference/cls \
--table_model_dir=inference/table \
--image_dir=../../ppstructure/docs/table/table.jpg \
- --use_angle_cls=true \
- --det=true \
- --rec=true \
- --cls=true \
--type=structure \
--table=true
```
From ec22e60cb4eeb8ee33d6c4ec2a7c888751efb3b0 Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Tue, 16 Aug 2022 03:34:40 +0000
Subject: [PATCH 26/49] fix boxlabel error
---
ppocr/data/imaug/label_ops.py | 2 +-
1 file changed, 1 insertion(+), 1 deletion(-)
diff --git a/ppocr/data/imaug/label_ops.py b/ppocr/data/imaug/label_ops.py
index 73d4ca430f..59cb9b8a25 100644
--- a/ppocr/data/imaug/label_ops.py
+++ b/ppocr/data/imaug/label_ops.py
@@ -750,7 +750,7 @@ class TableMasterLabelEncode(TableLabelEncode):
class TableBoxEncode(object):
def __init__(self, in_box_format='xyxy', out_box_format='xyxy', **kwargs):
- assert box_format in ['xywh', 'xyxy', 'xyxyxyxy']
+ assert out_box_format in ['xywh', 'xyxy', 'xyxyxyxy']
self.in_box_format = in_box_format
self.out_box_format = out_box_format
From 7327baf1644415937d473b09f14fbc7c25e3858e Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Tue, 16 Aug 2022 03:44:39 +0000
Subject: [PATCH 27/49] add en install doc
---
ppstructure/docs/installation_en.md | 30 +++++++++++++++++++++++++++++
1 file changed, 30 insertions(+)
create mode 100644 ppstructure/docs/installation_en.md
diff --git a/ppstructure/docs/installation_en.md b/ppstructure/docs/installation_en.md
new file mode 100644
index 0000000000..02b02db0c5
--- /dev/null
+++ b/ppstructure/docs/installation_en.md
@@ -0,0 +1,30 @@
+# Quick installation
+
+- [1. PaddlePaddle 和 PaddleOCR](#1)
+- [2. Install other dependencies](#2)
+ - [2.1 VQA](#21)
+
+
+
+## 1. PaddlePaddle and PaddleOCR
+
+Please refer to [PaddleOCR installation documentation](../../doc/doc_en/installation_en.md)
+
+
+## 2. Install other dependencies
+
+
+### 2.1 VQA
+
+* paddleocr
+
+```bash
+pip3 install paddleocr
+```
+
+* PaddleNLP
+```bash
+git clone https://github.com/PaddlePaddle/PaddleNLP -b develop
+cd PaddleNLP
+pip3 install -e .
+```
From 02e881e5080a32d1c9e6aa2f589c98edaf8cef61 Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Tue, 16 Aug 2022 07:05:33 +0000
Subject: [PATCH 28/49] add table en doc
---
ppstructure/table/README.md | 27 +++++++++++++--------------
1 file changed, 13 insertions(+), 14 deletions(-)
diff --git a/ppstructure/table/README.md b/ppstructure/table/README.md
index b6804c6f09..45c13565ee 100644
--- a/ppstructure/table/README.md
+++ b/ppstructure/table/README.md
@@ -32,7 +32,8 @@ We evaluated the algorithm on the PubTabNet[1] eval dataset, and the
|Method|[TEDS(Tree-Edit-Distance-based Similarity)](https://github.com/ibm-aur-nlp/PubTabNet/tree/master/src)|
| --- | --- |
| EDD[2] | 88.3 |
-| Ours | 93.32 |
+| TableRec-RARE(ours) | 93.32 |
+| SLANet(ours) | 94.98 |
## 3. How to use
@@ -55,7 +56,7 @@ python3 table/predict_table.py --det_model_dir=inference/en_ppocr_mobile_v2.0_ta
```
Note: The above model is trained on the PubLayNet dataset and only supports English scanning scenarios. If you need to identify other scenarios, you need to train the model yourself and replace the three fields `det_model_dir`, `rec_model_dir`, `table_model_dir`.
-After running, the excel sheet of each picture will be saved in the directory specified by the output field
+After the operation is completed, the excel table of each image will be saved to the directory specified by the output field, and an html file will be produced in the directory to visually view the cell coordinates and the recognized table.
### 3.2 Train
@@ -90,27 +91,25 @@ python3 tools/train.py -c configs/table/table_mv3.yml -o Global.checkpoints=./yo
### 3.3 Eval
The table uses [TEDS(Tree-Edit-Distance-based Similarity)](https://github.com/ibm-aur-nlp/PubTabNet/tree/master/src) as the evaluation metric of the model. Before the model evaluation, the three models in the pipeline need to be exported as inference models (we have provided them), and the gt for evaluation needs to be prepared. Examples of gt are as follows:
-```json
-{"PMC4289340_004_00.png": [
- ["", "", "", "", "", "| ", " | ", "", " | ", "", " | ", " ", "", "", "", "| ", " | ", "", " | ", "", " | ", " ", "", " ", "", ""],
- [[1, 4, 29, 13], [137, 4, 161, 13], [215, 4, 236, 13], [1, 17, 30, 27], [137, 17, 147, 27], [215, 17, 225, 27]],
- [["", "F", "e", "a", "t", "u", "r", "e", ""], ["", "G", "b", "3", " ", "+", ""], ["", "G", "b", "3", " ", "-", ""], ["", "P", "a", "t", "i", "e", "n", "t", "s", ""], ["6", "2"], ["4", "5"]]
-]}
+```txt
+PMC5755158_010_01.png | Weaning | Week 15 | Off-test | | Weaning | – | – | – | | Week 15 | – | 0.17 ± 0.08 | 0.16 ± 0.03 | | Off-test | – | 0.80 ± 0.24 | 0.19 ± 0.09 |
+```
+Each line in gt consists of the file name and the html string of the table. The file name and the html string of the table are separated by `\t`.
+
+You can also use the following command to generate an evaluation gt file from the annotation file:
+```python
+python3 ppstructure/table/convert_label2html.py --ori_gt_path /path/to/your_label_file --save_path /path/to/save_file
```
-In gt json, the key is the image name, the value is the corresponding gt, and gt is a list composed of four items, and each item is
-1. HTML string list of table structure
-2. The coordinates of each cell (not including the empty text in the cell)
-3. The text information in each cell (not including the empty text in the cell)
Use the following command to evaluate. After the evaluation is completed, the teds indicator will be output.
```python
cd PaddleOCR/ppstructure
-python3 table/eval_table.py --det_model_dir=path/to/det_model_dir --rec_model_dir=path/to/rec_model_dir --table_model_dir=path/to/table_model_dir --image_dir=../doc/table/1.png --rec_char_dict_path=../ppocr/utils/dict/table_dict.txt --table_char_dict_path=../ppocr/utils/dict/table_structure_dict.txt --det_limit_side_len=736 --det_limit_type=min --gt_path=path/to/gt.json
+python3 table/eval_table.py --det_model_dir=path/to/det_model_dir --rec_model_dir=path/to/rec_model_dir --table_model_dir=path/to/table_model_dir --image_dir=../doc/table/1.png --rec_char_dict_path=../ppocr/utils/dict/table_dict.txt --table_char_dict_path=../ppocr/utils/dict/table_structure_dict.txt --det_limit_side_len=736 --det_limit_type=min --gt_path=path/to/gt.txt
```
If the PubLatNet eval dataset is used, it will be output
```bash
-teds: 93.32
+teds: 94.98
```
### 3.4 Inference
From 9863be35f8b7ed7b8d5819fe988bed657db67563 Mon Sep 17 00:00:00 2001
From: zhengya01
Date: Tue, 16 Aug 2022 15:18:05 +0800
Subject: [PATCH 29/49] add log_path in result.log
---
test_tipc/common_func.sh | 5 +++--
test_tipc/test_inference_cpp.sh | 4 ++--
test_tipc/test_inference_python.sh | 9 +++++----
test_tipc/test_paddle2onnx.sh | 14 +++++++-------
test_tipc/test_ptq_inference_python.sh | 6 +++---
test_tipc/test_serving_infer_cpp.sh | 8 ++++----
test_tipc/test_serving_infer_python.sh | 16 ++++++++--------
test_tipc/test_train_inference_python.sh | 12 ++++++------
8 files changed, 38 insertions(+), 36 deletions(-)
diff --git a/test_tipc/common_func.sh b/test_tipc/common_func.sh
index f7d8a1e04a..1bbf829165 100644
--- a/test_tipc/common_func.sh
+++ b/test_tipc/common_func.sh
@@ -58,10 +58,11 @@ function status_check(){
run_command=$2
run_log=$3
model_name=$4
+ log_path=$5
if [ $last_status -eq 0 ]; then
- echo -e "\033[33m Run successfully with command - ${model_name} - ${run_command}! \033[0m" | tee -a ${run_log}
+ echo -e "\033[33m Run successfully with command - ${model_name} - ${run_command} - ${log_path} \033[0m" | tee -a ${run_log}
else
- echo -e "\033[33m Run failed with command - ${model_name} - ${run_command}! \033[0m" | tee -a ${run_log}
+ echo -e "\033[33m Run failed with command - ${model_name} - ${run_command} - ${log_path} \033[0m" | tee -a ${run_log}
fi
}
diff --git a/test_tipc/test_inference_cpp.sh b/test_tipc/test_inference_cpp.sh
index c0c7c18a38..aadaa8b077 100644
--- a/test_tipc/test_inference_cpp.sh
+++ b/test_tipc/test_inference_cpp.sh
@@ -84,7 +84,7 @@ function func_cpp_inference(){
eval $command
last_status=${PIPESTATUS[0]}
eval "cat ${_save_log_path}"
- status_check $last_status "${command}" "${status_log}" "${model_name}"
+ status_check $last_status "${command}" "${status_log}" "${model_name}" "${_save_log_path}"
done
done
done
@@ -117,7 +117,7 @@ function func_cpp_inference(){
eval $command
last_status=${PIPESTATUS[0]}
eval "cat ${_save_log_path}"
- status_check $last_status "${command}" "${status_log}" "${model_name}"
+ status_check $last_status "${command}" "${status_log}" "${model_name}" "${_save_log_path}"
done
done
diff --git a/test_tipc/test_inference_python.sh b/test_tipc/test_inference_python.sh
index 2a31a468f0..e9908df1f6 100644
--- a/test_tipc/test_inference_python.sh
+++ b/test_tipc/test_inference_python.sh
@@ -88,7 +88,7 @@ function func_inference(){
eval $command
last_status=${PIPESTATUS[0]}
eval "cat ${_save_log_path}"
- status_check $last_status "${command}" "${status_log}" "${model_name}"
+ status_check $last_status "${command}" "${status_log}" "${model_name}" "${_save_log_path}"
done
done
done
@@ -119,7 +119,7 @@ function func_inference(){
eval $command
last_status=${PIPESTATUS[0]}
eval "cat ${_save_log_path}"
- status_check $last_status "${command}" "${status_log}" "${model_name}"
+ status_check $last_status "${command}" "${status_log}" "${model_name}" "${_save_log_path}"
done
done
@@ -146,14 +146,15 @@ if [ ${MODE} = "whole_infer" ]; then
for infer_model in ${infer_model_dir_list[*]}; do
# run export
if [ ${infer_run_exports[Count]} != "null" ];then
+ _save_log_path="${_log_path}/python_infer_gpu_usetrt_${use_trt}_precision_${precision}_batchsize_${batch_size}_infermodel_${infer_model}.log"
save_infer_dir=$(dirname $infer_model)
set_export_weight=$(func_set_params "${export_weight}" "${infer_model}")
set_save_infer_key=$(func_set_params "${save_infer_key}" "${save_infer_dir}")
- export_cmd="${python} ${infer_run_exports[Count]} ${set_export_weight} ${set_save_infer_key}"
+ export_cmd="${python} ${infer_run_exports[Count]} ${set_export_weight} ${set_save_infer_key} > ${_save_log_path} 2>&1 "
echo ${infer_run_exports[Count]}
eval $export_cmd
status_export=$?
- status_check $status_export "${export_cmd}" "${status_log}" "${model_name}"
+ status_check $status_export "${export_cmd}" "${status_log}" "${model_name}" "${_save_log_path}"
else
save_infer_dir=${infer_model}
fi
diff --git a/test_tipc/test_paddle2onnx.sh b/test_tipc/test_paddle2onnx.sh
index 78d79d0b8e..bace6b2d46 100644
--- a/test_tipc/test_paddle2onnx.sh
+++ b/test_tipc/test_paddle2onnx.sh
@@ -66,7 +66,7 @@ function func_paddle2onnx(){
trans_model_cmd="${padlle2onnx_cmd} ${set_dirname} ${set_model_filename} ${set_params_filename} ${set_save_model} ${set_opset_version} ${set_enable_onnx_checker} > ${trans_det_log} 2>&1 "
eval $trans_model_cmd
last_status=${PIPESTATUS[0]}
- status_check $last_status "${trans_model_cmd}" "${status_log}" "${model_name}"
+ status_check $last_status "${trans_model_cmd}" "${status_log}" "${model_name}" "${trans_det_log}"
# trans rec
set_dirname=$(func_set_params "--model_dir" "${rec_infer_model_dir_value}")
set_model_filename=$(func_set_params "${model_filename_key}" "${model_filename_value}")
@@ -78,7 +78,7 @@ function func_paddle2onnx(){
trans_model_cmd="${padlle2onnx_cmd} ${set_dirname} ${set_model_filename} ${set_params_filename} ${set_save_model} ${set_opset_version} ${set_enable_onnx_checker} > ${trans_rec_log} 2>&1 "
eval $trans_model_cmd
last_status=${PIPESTATUS[0]}
- status_check $last_status "${trans_model_cmd}" "${status_log}" "${model_name}"
+ status_check $last_status "${trans_model_cmd}" "${status_log}" "${model_name}" "${trans_rec_log}"
elif [[ ${model_name} =~ "det" ]]; then
# trans det
set_dirname=$(func_set_params "--model_dir" "${det_infer_model_dir_value}")
@@ -91,7 +91,7 @@ function func_paddle2onnx(){
trans_model_cmd="${padlle2onnx_cmd} ${set_dirname} ${set_model_filename} ${set_params_filename} ${set_save_model} ${set_opset_version} ${set_enable_onnx_checker} > ${trans_det_log} 2>&1 "
eval $trans_model_cmd
last_status=${PIPESTATUS[0]}
- status_check $last_status "${trans_model_cmd}" "${status_log}" "${model_name}"
+ status_check $last_status "${trans_model_cmd}" "${status_log}" "${model_name}" "${trans_det_log}"
elif [[ ${model_name} =~ "rec" ]]; then
# trans rec
set_dirname=$(func_set_params "--model_dir" "${rec_infer_model_dir_value}")
@@ -104,7 +104,7 @@ function func_paddle2onnx(){
trans_model_cmd="${padlle2onnx_cmd} ${set_dirname} ${set_model_filename} ${set_params_filename} ${set_save_model} ${set_opset_version} ${set_enable_onnx_checker} > ${trans_rec_log} 2>&1 "
eval $trans_model_cmd
last_status=${PIPESTATUS[0]}
- status_check $last_status "${trans_model_cmd}" "${status_log}" "${model_name}"
+ status_check $last_status "${trans_model_cmd}" "${status_log}" "${model_name}" "${trans_rec_log}"
fi
# python inference
@@ -127,7 +127,7 @@ function func_paddle2onnx(){
eval $infer_model_cmd
last_status=${PIPESTATUS[0]}
eval "cat ${_save_log_path}"
- status_check $last_status "${infer_model_cmd}" "${status_log}" "${model_name}"
+ status_check $last_status "${infer_model_cmd}" "${status_log}" "${model_name}" "${_save_log_path}"
elif [ ${use_gpu} = "True" ] || [ ${use_gpu} = "gpu" ]; then
_save_log_path="${LOG_PATH}/paddle2onnx_infer_gpu.log"
set_gpu=$(func_set_params "${use_gpu_key}" "${use_gpu}")
@@ -146,7 +146,7 @@ function func_paddle2onnx(){
eval $infer_model_cmd
last_status=${PIPESTATUS[0]}
eval "cat ${_save_log_path}"
- status_check $last_status "${infer_model_cmd}" "${status_log}" "${model_name}"
+ status_check $last_status "${infer_model_cmd}" "${status_log}" "${model_name}" "${_save_log_path}"
else
echo "Does not support hardware other than CPU and GPU Currently!"
fi
@@ -158,4 +158,4 @@ echo "################### run test ###################"
export Count=0
IFS="|"
-func_paddle2onnx
\ No newline at end of file
+func_paddle2onnx
diff --git a/test_tipc/test_ptq_inference_python.sh b/test_tipc/test_ptq_inference_python.sh
index e2939fd5e6..caf3d50602 100644
--- a/test_tipc/test_ptq_inference_python.sh
+++ b/test_tipc/test_ptq_inference_python.sh
@@ -84,7 +84,7 @@ function func_inference(){
eval $command
last_status=${PIPESTATUS[0]}
eval "cat ${_save_log_path}"
- status_check $last_status "${command}" "${status_log}" "${model_name}"
+ status_check $last_status "${command}" "${status_log}" "${model_name}" "${_save_log_path}"
done
done
done
@@ -109,7 +109,7 @@ function func_inference(){
eval $command
last_status=${PIPESTATUS[0]}
eval "cat ${_save_log_path}"
- status_check $last_status "${command}" "${status_log}" "${model_name}"
+ status_check $last_status "${command}" "${status_log}" "${model_name}" "${_save_log_path}"
done
done
@@ -145,7 +145,7 @@ if [ ${MODE} = "whole_infer" ]; then
echo $export_cmd
eval $export_cmd
status_export=$?
- status_check $status_export "${export_cmd}" "${status_log}" "${model_name}"
+ status_check $status_export "${export_cmd}" "${status_log}" "${model_name}" "${export_log_path}"
else
save_infer_dir=${infer_model}
fi
diff --git a/test_tipc/test_serving_infer_cpp.sh b/test_tipc/test_serving_infer_cpp.sh
index 0be6a45adf..10ddecf3fa 100644
--- a/test_tipc/test_serving_infer_cpp.sh
+++ b/test_tipc/test_serving_infer_cpp.sh
@@ -83,7 +83,7 @@ function func_serving(){
trans_model_cmd="${python_list[0]} ${trans_model_py} ${set_dirname} ${set_model_filename} ${set_params_filename} ${set_serving_server} ${set_serving_client} > ${trans_rec_log} 2>&1 "
eval $trans_model_cmd
last_status=${PIPESTATUS[0]}
- status_check $last_status "${trans_model_cmd}" "${status_log}" "${model_name}"
+ status_check $last_status "${trans_model_cmd}" "${status_log}" "${model_name}" "${trans_rec_log}"
set_image_dir=$(func_set_params "${image_dir_key}" "${image_dir_value}")
python_list=(${python_list})
cd ${serving_dir_value}
@@ -95,14 +95,14 @@ function func_serving(){
web_service_cpp_cmd="nohup ${python_list[0]} ${web_service_py} --model ${det_server_value} ${rec_server_value} ${op_key} ${op_value} ${port_key} ${port_value} > ${server_log_path} 2>&1 &"
eval $web_service_cpp_cmd
last_status=${PIPESTATUS[0]}
- status_check $last_status "${web_service_cpp_cmd}" "${status_log}" "${model_name}"
+ status_check $last_status "${web_service_cpp_cmd}" "${status_log}" "${model_name}" "${server_log_path}"
sleep 5s
_save_log_path="${LOG_PATH}/cpp_client_cpu.log"
cpp_client_cmd="${python_list[0]} ${cpp_client_py} ${det_client_value} ${rec_client_value} > ${_save_log_path} 2>&1"
eval $cpp_client_cmd
last_status=${PIPESTATUS[0]}
eval "cat ${_save_log_path}"
- status_check $last_status "${cpp_client_cmd}" "${status_log}" "${model_name}"
+ status_check $last_status "${cpp_client_cmd}" "${status_log}" "${model_name}" "${_save_log_path}"
ps ux | grep -i ${port_value} | awk '{print $2}' | xargs kill -s 9
else
server_log_path="${LOG_PATH}/cpp_server_gpu.log"
@@ -114,7 +114,7 @@ function func_serving(){
eval $cpp_client_cmd
last_status=${PIPESTATUS[0]}
eval "cat ${_save_log_path}"
- status_check $last_status "${cpp_client_cmd}" "${status_log}" "${model_name}"
+ status_check $last_status "${cpp_client_cmd}" "${status_log}" "${model_name}" "${_save_log_path}"
ps ux | grep -i ${port_value} | awk '{print $2}' | xargs kill -s 9
fi
done
diff --git a/test_tipc/test_serving_infer_python.sh b/test_tipc/test_serving_infer_python.sh
index 4b7dfcf785..c7d305d5d2 100644
--- a/test_tipc/test_serving_infer_python.sh
+++ b/test_tipc/test_serving_infer_python.sh
@@ -126,19 +126,19 @@ function func_serving(){
web_service_cmd="nohup ${python} ${web_service_py} ${web_use_gpu_key}="" ${web_use_mkldnn_key}=${use_mkldnn} ${set_cpu_threads} ${set_det_model_config} ${set_rec_model_config} > ${server_log_path} 2>&1 &"
eval $web_service_cmd
last_status=${PIPESTATUS[0]}
- status_check $last_status "${web_service_cmd}" "${status_log}" "${model_name}"
+ status_check $last_status "${web_service_cmd}" "${status_log}" "${model_name}" "${server_log_path}"
elif [[ ${model_name} =~ "det" ]]; then
set_det_model_config=$(func_set_params "${det_server_key}" "${det_server_value}")
web_service_cmd="nohup ${python} ${web_service_py} ${web_use_gpu_key}="" ${web_use_mkldnn_key}=${use_mkldnn} ${set_cpu_threads} ${set_det_model_config} > ${server_log_path} 2>&1 &"
eval $web_service_cmd
last_status=${PIPESTATUS[0]}
- status_check $last_status "${web_service_cmd}" "${status_log}" "${model_name}"
+ status_check $last_status "${web_service_cmd}" "${status_log}" "${model_name}" "${server_log_path}"
elif [[ ${model_name} =~ "rec" ]]; then
set_rec_model_config=$(func_set_params "${rec_server_key}" "${rec_server_value}")
web_service_cmd="nohup ${python} ${web_service_py} ${web_use_gpu_key}="" ${web_use_mkldnn_key}=${use_mkldnn} ${set_cpu_threads} ${set_rec_model_config} > ${server_log_path} 2>&1 &"
eval $web_service_cmd
last_status=${PIPESTATUS[0]}
- status_check $last_status "${web_service_cmd}" "${status_log}" "${model_name}"
+ status_check $last_status "${web_service_cmd}" "${status_log}" "${model_name}" "${server_log_path}"
fi
sleep 2s
for pipeline in ${pipeline_py[*]}; do
@@ -147,7 +147,7 @@ function func_serving(){
eval $pipeline_cmd
last_status=${PIPESTATUS[0]}
eval "cat ${_save_log_path}"
- status_check $last_status "${pipeline_cmd}" "${status_log}" "${model_name}"
+ status_check $last_status "${pipeline_cmd}" "${status_log}" "${model_name}" "${_save_log_path}"
sleep 2s
done
ps ux | grep -E 'web_service' | awk '{print $2}' | xargs kill -s 9
@@ -177,19 +177,19 @@ function func_serving(){
web_service_cmd="nohup ${python} ${web_service_py} ${set_tensorrt} ${set_precision} ${set_det_model_config} ${set_rec_model_config} > ${server_log_path} 2>&1 &"
eval $web_service_cmd
last_status=${PIPESTATUS[0]}
- status_check $last_status "${web_service_cmd}" "${status_log}" "${model_name}"
+ status_check $last_status "${web_service_cmd}" "${status_log}" "${model_name}" "${server_log_path}"
elif [[ ${model_name} =~ "det" ]]; then
set_det_model_config=$(func_set_params "${det_server_key}" "${det_server_value}")
web_service_cmd="nohup ${python} ${web_service_py} ${set_tensorrt} ${set_precision} ${set_det_model_config} > ${server_log_path} 2>&1 &"
eval $web_service_cmd
last_status=${PIPESTATUS[0]}
- status_check $last_status "${web_service_cmd}" "${status_log}" "${model_name}"
+ status_check $last_status "${web_service_cmd}" "${status_log}" "${model_name}" "${server_log_path}"
elif [[ ${model_name} =~ "rec" ]]; then
set_rec_model_config=$(func_set_params "${rec_server_key}" "${rec_server_value}")
web_service_cmd="nohup ${python} ${web_service_py} ${set_tensorrt} ${set_precision} ${set_rec_model_config} > ${server_log_path} 2>&1 &"
eval $web_service_cmd
last_status=${PIPESTATUS[0]}
- status_check $last_status "${web_service_cmd}" "${status_log}" "${model_name}"
+ status_check $last_status "${web_service_cmd}" "${status_log}" "${model_name}" "${server_log_path}"
fi
sleep 2s
for pipeline in ${pipeline_py[*]}; do
@@ -198,7 +198,7 @@ function func_serving(){
eval $pipeline_cmd
last_status=${PIPESTATUS[0]}
eval "cat ${_save_log_path}"
- status_check $last_status "${pipeline_cmd}" "${status_log}" "${model_name}"
+ status_check $last_status "${pipeline_cmd}" "${status_log}" "${model_name}" "${_save_log_path}"
sleep 2s
done
ps ux | grep -E 'web_service' | awk '{print $2}' | xargs kill -s 9
diff --git a/test_tipc/test_train_inference_python.sh b/test_tipc/test_train_inference_python.sh
index 545cdbba20..e182fa57f0 100644
--- a/test_tipc/test_train_inference_python.sh
+++ b/test_tipc/test_train_inference_python.sh
@@ -133,7 +133,7 @@ function func_inference(){
eval $command
last_status=${PIPESTATUS[0]}
eval "cat ${_save_log_path}"
- status_check $last_status "${command}" "${status_log}" "${model_name}"
+ status_check $last_status "${command}" "${status_log}" "${model_name}" "${_save_log_path}"
done
done
done
@@ -164,7 +164,7 @@ function func_inference(){
eval $command
last_status=${PIPESTATUS[0]}
eval "cat ${_save_log_path}"
- status_check $last_status "${command}" "${status_log}" "${model_name}"
+ status_check $last_status "${command}" "${status_log}" "${model_name}" "${_save_log_path}"
done
done
@@ -201,7 +201,7 @@ if [ ${MODE} = "whole_infer" ]; then
echo $export_cmd
eval $export_cmd
status_export=$?
- status_check $status_export "${export_cmd}" "${status_log}" "${model_name}"
+ status_check $status_export "${export_cmd}" "${status_log}" "${model_name}" "${export_log_path}"
else
save_infer_dir=${infer_model}
fi
@@ -298,7 +298,7 @@ else
# run train
eval $cmd
eval "cat ${save_log}/train.log >> ${save_log}.log"
- status_check $? "${cmd}" "${status_log}" "${model_name}"
+ status_check $? "${cmd}" "${status_log}" "${model_name}" "${save_log}.log"
set_eval_pretrain=$(func_set_params "${pretrain_model_key}" "${save_log}/${train_model_name}")
@@ -309,7 +309,7 @@ else
eval_log_path="${LOG_PATH}/${trainer}_gpus_${gpu}_autocast_${autocast}_nodes_${nodes}_eval.log"
eval_cmd="${python} ${eval_py} ${set_eval_pretrain} ${set_use_gpu} ${set_eval_params1} > ${eval_log_path} 2>&1 "
eval $eval_cmd
- status_check $? "${eval_cmd}" "${status_log}" "${model_name}"
+ status_check $? "${eval_cmd}" "${status_log}" "${model_name}" "${eval_log_path}"
fi
# run export model
if [ ${run_export} != "null" ]; then
@@ -320,7 +320,7 @@ else
set_save_infer_key=$(func_set_params "${save_infer_key}" "${save_infer_path}")
export_cmd="${python} ${run_export} ${set_export_weight} ${set_save_infer_key} > ${export_log_path} 2>&1 "
eval $export_cmd
- status_check $? "${export_cmd}" "${status_log}" "${model_name}"
+ status_check $? "${export_cmd}" "${status_log}" "${model_name}" "${export_log_path}"
#run inference
eval $env
From d69b74e4345e57ccb615e2132a9d414ac1721940 Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Tue, 16 Aug 2022 07:45:51 +0000
Subject: [PATCH 30/49] add dataset desc
---
doc/doc_ch/dataset/table_datasets.md | 11 +++++++++++
doc/doc_en/dataset/table_datasets_en.md | 10 ++++++++++
ppstructure/table/README.md | 11 ++++++++++-
ppstructure/table/README_ch.md | 10 +++++++++-
4 files changed, 40 insertions(+), 2 deletions(-)
diff --git a/doc/doc_ch/dataset/table_datasets.md b/doc/doc_ch/dataset/table_datasets.md
index ae902b23cc..58f4cf4705 100644
--- a/doc/doc_ch/dataset/table_datasets.md
+++ b/doc/doc_ch/dataset/table_datasets.md
@@ -3,6 +3,7 @@
- [数据集汇总](#数据集汇总)
- [1. PubTabNet数据集](#1-pubtabnet数据集)
- [2. 好未来表格识别竞赛数据集](#2-好未来表格识别竞赛数据集)
+- [3. 好未来表格识别竞赛数据集](#2-WTW中文场景表格数据集)
这里整理了常用表格识别数据集,持续更新中,欢迎各位小伙伴贡献数据集~
@@ -12,6 +13,7 @@
|---|---|---|
| PubTabNet |https://github.com/ibm-aur-nlp/PubTabNet| jsonl格式,可直接用[pubtab_dataset.py](../../../ppocr/data/pubtab_dataset.py)加载 |
| 好未来表格识别竞赛数据集 |https://ai.100tal.com/dataset| jsonl格式,可直接用[pubtab_dataset.py](../../../ppocr/data/pubtab_dataset.py)加载 |
+| WTW中文场景表格数据集 |https://github.com/wangwen-whu/WTW-Dataset| 需要进行转换后才能用[pubtab_dataset.py](../../../ppocr/data/pubtab_dataset.py)加载 |
## 1. PubTabNet数据集
- **数据简介**:PubTabNet数据集的训练集合中包含50万张图像,验证集合中包含0.9万张图像。部分图像可视化如下所示。
@@ -31,3 +33,12 @@
+
+## 3. WTW中文场景表格数据集
+- **数据简介**:WTW中文场景表格数据集包含表格检测和表格数据两部分数据,数据集中同时包含扫描和拍照两张场景的图像。
+
+https://github.com/wangwen-whu/WTW-Dataset/blob/main/demo/20210816_210413.gif
+
+
+ 
+
diff --git a/doc/doc_en/dataset/table_datasets_en.md b/doc/doc_en/dataset/table_datasets_en.md
index e301479098..70ca830979 100644
--- a/doc/doc_en/dataset/table_datasets_en.md
+++ b/doc/doc_en/dataset/table_datasets_en.md
@@ -3,6 +3,7 @@
- [Dataset Summary](#dataset-summary)
- [1. PubTabNet](#1-pubtabnet)
- [2. TAL Table Recognition Competition Dataset](#2-tal-table-recognition-competition-dataset)
+- [3. WTW Chinese scene table dataset](#3-wtw-chinese-scene-table-dataset)
Here are the commonly used table recognition datasets, which are being updated continuously. Welcome to contribute datasets~
@@ -12,6 +13,7 @@ Here are the commonly used table recognition datasets, which are being updated c
|---|---|---|
| PubTabNet |https://github.com/ibm-aur-nlp/PubTabNet| jsonl format, which can be loaded directly with [pubtab_dataset.py](../../../ppocr/data/pubtab_dataset.py) |
| TAL Table Recognition Competition Dataset |https://ai.100tal.com/dataset| jsonl format, which can be loaded directly with [pubtab_dataset.py](../../../ppocr/data/pubtab_dataset.py) |
+| WTW Chinese scene table dataset |https://github.com/wangwen-whu/WTW-Dataset| Conversion is required to load with [pubtab_dataset.py](../../../ppocr/data/pubtab_dataset.py)|
## 1. PubTabNet
- **Data Introduction**:The training set of the PubTabNet dataset contains 500,000 images and the validation set contains 9000 images. Part of the image visualization is shown below.
@@ -30,3 +32,11 @@ Here are the commonly used table recognition datasets, which are being updated c
+
+## 3. WTW Chinese scene table dataset
+- **Data Introduction**:The WTW Chinese scene table dataset consists of two parts: table detection and table data. The dataset contains images of two scenes, scanned and photographed.
+https://github.com/wangwen-whu/WTW-Dataset/blob/main/demo/20210816_210413.gif
+
+
+ 
+
diff --git a/ppstructure/table/README.md b/ppstructure/table/README.md
index 45c13565ee..10308b4923 100644
--- a/ppstructure/table/README.md
+++ b/ppstructure/table/README.md
@@ -63,7 +63,16 @@ After the operation is completed, the excel table of each image will be saved to
In this chapter, we only introduce the training of the table structure model, For model training of [text detection](../../doc/doc_en/detection_en.md) and [text recognition](../../doc/doc_en/recognition_en.md), please refer to the corresponding documents
* data preparation
-The training data uses public data set [PubTabNet](https://arxiv.org/abs/1911.10683 ), Can be downloaded from the official [website](https://github.com/ibm-aur-nlp/PubTabNet) 。The PubTabNet data set contains about 500,000 images, as well as annotations in html format。
+
+For the Chinese model and the English model, the data sources are different, as follows:
+
+English dataset: The training data uses public data set [PubTabNet](https://arxiv.org/abs/1911.10683 ), Can be downloaded from the official [website](https://github.com/ibm-aur-nlp/PubTabNet) 。The PubTabNet data set contains about 500,000 images, as well as annotations in html format。
+
+Chinese dataset: The Chinese dataset consists of the following two parts, which are trained with a 1:1 sampling ratio.
+> 1. Generate dataset: Use [Table Generation Tool](https://github.com/WenmuZhou/TableGeneration) to generate 40,000 images.
+> 2. Crop 10,000 images from [WTW](https://github.com/wangwen-whu/WTW-Dataset).
+
+For a detailed introduction to public datasets, please refer to [table_datasets](../../doc/doc_en/dataset/table_datasets_en.md). The following training and evaluation procedures are based on the English dataset as an example.
* Start training
*If you are installing the cpu version of paddle, please modify the `use_gpu` field in the configuration file to false*
diff --git a/ppstructure/table/README_ch.md b/ppstructure/table/README_ch.md
index 21fb7960cc..3f31c0106b 100644
--- a/ppstructure/table/README_ch.md
+++ b/ppstructure/table/README_ch.md
@@ -75,7 +75,15 @@ note: 上述模型是在 PubLayNet 数据集上训练的表格识别模型,仅
* 数据准备
-训练数据使用公开数据集PubTabNet ([论文](https://arxiv.org/abs/1911.10683),[下载地址](https://github.com/ibm-aur-nlp/PubTabNet))。PubTabNet数据集包含约50万张表格数据的图像,以及图像对应的html格式的注释。
+对于中文模型和英文模型,数据来源不同,分别介绍如下
+
+英文数据集: 训练数据使用公开数据集PubTabNet ([论文](https://arxiv.org/abs/1911.10683),[下载地址](https://github.com/ibm-aur-nlp/PubTabNet))。PubTabNet数据集包含约50万张表格数据的图像,以及图像对应的html格式的注释。
+
+中文数据集: 中文数据集下面两部分构成,这两部分安装1:1的采样比例进行训练。
+> 1. 生成数据集: 使用[表格生成工具](https://github.com/WenmuZhou/TableGeneration)生成4w张。
+> 2. 从[WTW](https://github.com/wangwen-whu/WTW-Dataset)中获取1w张。
+
+关于公开数据集的详细介绍可以参考 [table_datasets](../../doc/doc_ch/dataset/table_datasets.md),下述训练和评估流程均以英文数据集为例。
* 启动训练
From 6e89ec8d09c06453edeee3874a826e750a6947d6 Mon Sep 17 00:00:00 2001
From: andyjpaddle
Date: Tue, 16 Aug 2022 09:34:48 +0000
Subject: [PATCH 31/49] fix sar export
---
doc/doc_ch/algorithm_rec_sar.md | 2 +-
doc/doc_en/algorithm_rec_sar_en.md | 2 +-
tools/export_model.py | 4 +++-
tools/infer/predict_rec.py | 3 ++-
4 files changed, 7 insertions(+), 4 deletions(-)
diff --git a/doc/doc_ch/algorithm_rec_sar.md b/doc/doc_ch/algorithm_rec_sar.md
index b830431399..cfb1de2539 100644
--- a/doc/doc_ch/algorithm_rec_sar.md
+++ b/doc/doc_ch/algorithm_rec_sar.md
@@ -79,7 +79,7 @@ python3 tools/export_model.py -c configs/rec/rec_r31_sar.yml -o Global.pretraine
SAR文本识别模型推理,可以执行如下命令:
```
-python3 tools/infer/predict_rec.py --image_dir="./doc/imgs_words/en/word_1.png" --rec_model_dir="./inference/rec_sar/" --rec_image_shape="3, 48, 48, 160" --rec_char_type="ch" --rec_algorithm="SAR" --rec_char_dict_path="ppocr/utils/dict90.txt" --max_text_length=30 --use_space_char=False
+python3 tools/infer/predict_rec.py --image_dir="./doc/imgs_words/en/word_1.png" --rec_model_dir="./inference/rec_sar/" --rec_image_shape="3, 48, 48, 160" --rec_algorithm="SAR" --rec_char_dict_path="ppocr/utils/dict90.txt" --max_text_length=30 --use_space_char=False
```
diff --git a/doc/doc_en/algorithm_rec_sar_en.md b/doc/doc_en/algorithm_rec_sar_en.md
index 24b87c10c3..5c8319da3b 100644
--- a/doc/doc_en/algorithm_rec_sar_en.md
+++ b/doc/doc_en/algorithm_rec_sar_en.md
@@ -79,7 +79,7 @@ python3 tools/export_model.py -c configs/rec/rec_r31_sar.yml -o Global.pretraine
For SAR text recognition model inference, the following commands can be executed:
```
-python3 tools/infer/predict_rec.py --image_dir="./doc/imgs_words/en/word_1.png" --rec_model_dir="./inference/rec_sar/" --rec_image_shape="3, 48, 48, 160" --rec_char_type="ch" --rec_algorithm="SAR" --rec_char_dict_path="ppocr/utils/dict90.txt" --max_text_length=30 --use_space_char=False
+python3 tools/infer/predict_rec.py --image_dir="./doc/imgs_words/en/word_1.png" --rec_model_dir="./inference/rec_sar/" --rec_image_shape="3, 48, 48, 160" --rec_algorithm="SAR" --rec_char_dict_path="ppocr/utils/dict90.txt" --max_text_length=30 --use_space_char=False
```
diff --git a/tools/export_model.py b/tools/export_model.py
index a61e5ca934..54edb50413 100755
--- a/tools/export_model.py
+++ b/tools/export_model.py
@@ -58,6 +58,8 @@ def export_single_model(model,
other_shape = [
paddle.static.InputSpec(
shape=[None, 3, 48, 160], dtype="float32"),
+ [paddle.static.InputSpec(
+ shape=[None], dtype="float32")]
]
model = to_static(model, input_spec=other_shape)
elif arch_config["algorithm"] == "SVTR":
@@ -232,4 +234,4 @@ def main():
if __name__ == "__main__":
- main()
\ No newline at end of file
+ main()
diff --git a/tools/infer/predict_rec.py b/tools/infer/predict_rec.py
index 449f69ed6a..1a483da751 100755
--- a/tools/infer/predict_rec.py
+++ b/tools/infer/predict_rec.py
@@ -439,7 +439,8 @@ class TextRecognizer(object):
valid_ratios = np.concatenate(valid_ratios)
inputs = [
norm_img_batch,
- valid_ratios,
+ np.array(
+ [valid_ratios], dtype=np.float32),
]
if self.use_onnx:
input_dict = {}
From bb53c8d1002197a447ea3d8f7c5c2e4044267d39 Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Tue, 16 Aug 2022 10:46:09 +0000
Subject: [PATCH 32/49] add table model link
---
configs/table/SLANet.yml | 2 +-
configs/table/table_mv3.yml | 4 +-
paddleocr.py | 24 ++++++++--
ppocr/utils/dict/table_structure_dict_ch.txt | 48 ++++++++++++++++++++
ppstructure/docs/models_list.md | 4 +-
ppstructure/docs/models_list_en.md | 4 +-
ppstructure/table/README.md | 23 ++++++----
ppstructure/table/README_ch.md | 23 ++++++----
ppstructure/utility.py | 2 +-
9 files changed, 107 insertions(+), 27 deletions(-)
create mode 100644 ppocr/utils/dict/table_structure_dict_ch.txt
diff --git a/configs/table/SLANet.yml b/configs/table/SLANet.yml
index 2264eb14d2..384c95852e 100644
--- a/configs/table/SLANet.yml
+++ b/configs/table/SLANet.yml
@@ -61,7 +61,7 @@ Loss:
PostProcess:
name: TableLabelDecode
- merge_no_span_structure: &merge_no_span_structure False
+ merge_no_span_structure: &merge_no_span_structure True
Metric:
name: TableMetric
diff --git a/configs/table/table_mv3.yml b/configs/table/table_mv3.yml
index 87cda7db21..16c1457442 100755
--- a/configs/table/table_mv3.yml
+++ b/configs/table/table_mv3.yml
@@ -96,8 +96,8 @@ Train:
Eval:
dataset:
name: PubTabDataSet
- data_dir: /home/zhoujun20/table/PubTabNe/pubtabnet/val/
- label_file_list: [/home/zhoujun20/table/PubTabNe/pubtabnet/val_500.jsonl]
+ data_dir: train_data/table/pubtabnet/val/
+ label_file_list: [train_data/table/pubtabnet/PubTabNet_2.0.0_val.jsonl]
transforms:
- DecodeImage: # load image
img_mode: BGR
diff --git a/paddleocr.py b/paddleocr.py
index 9a9958abef..fb1427b83f 100644
--- a/paddleocr.py
+++ b/paddleocr.py
@@ -275,12 +275,14 @@ MODEL_URLS = {
'PP-Structurev2': {
'table': {
'en': {
- 'url': '',
+ 'url':
+ 'https://paddleocr.bj.bcebos.com/ppstructure/models/slanet/en_ppstructure_mobile_v2.0_SLANet_infer.tar',
'dict_path': 'ppocr/utils/dict/table_structure_dict.txt'
},
'ch': {
- 'url': '',
- 'dict_path': 'ppocr/utils/dict/table_structure_dict.txt'
+ 'url':
+ 'https://paddleocr.bj.bcebos.com/ppstructure/models/slanet/ch_ppstructure_mobile_v2.0_SLANet_infer.tar',
+ 'dict_path': 'ppocr/utils/dict/table_structure_dict_ch.txt'
}
},
'layout': {
@@ -565,7 +567,6 @@ class PPStructure(StructureSystem):
if params.layout_dict_path is None:
params.layout_dict_path = str(
Path(__file__).parent / layout_model_config['dict_path'])
-
logger.debug(params)
super().__init__(params)
@@ -628,3 +629,18 @@ def main():
for item in result:
item.pop('img')
logger.info(item)
+
+
+if __name__ == "__main__":
+ table_engine = PPStructure(layout=False, show_log=True)
+
+ save_folder = './output'
+ img_path = 'ppstructure/docs/table/table.jpg'
+ img = cv2.imread(img_path)
+ result = table_engine(img)
+ save_structure_res(result, save_folder,
+ os.path.basename(img_path).split('.')[0])
+
+ for line in result:
+ line.pop('img')
+ print(line)
diff --git a/ppocr/utils/dict/table_structure_dict_ch.txt b/ppocr/utils/dict/table_structure_dict_ch.txt
new file mode 100644
index 0000000000..0c59c0e999
--- /dev/null
+++ b/ppocr/utils/dict/table_structure_dict_ch.txt
@@ -0,0 +1,48 @@
+
+
+
+
+
+
+
+ |
+ |
+ colspan="2"
+ colspan="3"
+ colspan="4"
+ colspan="5"
+ colspan="6"
+ colspan="7"
+ colspan="8"
+ colspan="9"
+ colspan="10"
+ colspan="11"
+ colspan="12"
+ colspan="13"
+ colspan="14"
+ colspan="15"
+ colspan="16"
+ colspan="17"
+ colspan="18"
+ colspan="19"
+ colspan="20"
+ rowspan="2"
+ rowspan="3"
+ rowspan="4"
+ rowspan="5"
+ rowspan="6"
+ rowspan="7"
+ rowspan="8"
+ rowspan="9"
+ rowspan="10"
+ rowspan="11"
+ rowspan="12"
+ rowspan="13"
+ rowspan="14"
+ rowspan="15"
+ rowspan="16"
+ rowspan="17"
+ rowspan="18"
+ rowspan="19"
+ rowspan="20"
diff --git a/ppstructure/docs/models_list.md b/ppstructure/docs/models_list.md
index 89fa98d3b7..ef2994cabe 100644
--- a/ppstructure/docs/models_list.md
+++ b/ppstructure/docs/models_list.md
@@ -34,7 +34,9 @@
|模型名称|模型简介|推理模型大小|下载地址|
| --- | --- | --- | --- |
-|en_ppocr_mobile_v2.0_table_structure|PubTabNet数据集训练的英文表格场景的表格结构预测|18.6M|[推理模型](https://paddleocr.bj.bcebos.com/dygraph_v2.0/table/en_ppocr_mobile_v2.0_table_structure_infer.tar) / [训练模型](https://paddleocr.bj.bcebos.com/dygraph_v2.1/table/en_ppocr_mobile_v2.0_table_structure_train.tar) |
+|en_ppocr_mobile_v2.0_table_structure|基于TableRec-RARE在PubTabNet数据集上训练的英文表格识别模型|18.6M|[推理模型](https://paddleocr.bj.bcebos.com/dygraph_v2.0/table/en_ppocr_mobile_v2.0_table_structure_infer.tar) / [训练模型](https://paddleocr.bj.bcebos.com/dygraph_v2.1/table/en_ppocr_mobile_v2.0_table_structure_train.tar) |
+|en_ppstructure_mobile_v2.0_SLANet|基于SLANet在PubTabNet数据集上训练的英文表格识别模型|9M|[推理模型](https://paddleocr.bj.bcebos.com/ppstructure/models/slanet/en_ppstructure_mobile_v2.0_SLANet_infer.tar) / [训练模型](https://paddleocr.bj.bcebos.com/ppstructure/models/slanet/en_ppstructure_mobile_v2.0_SLANet_train.tar) |
+|ch_ppstructure_mobile_v2.0_SLANet|基于SLANet在PubTabNet数据集上训练的中文表格识别模型|9.3M|[推理模型](https://paddleocr.bj.bcebos.com/ppstructure/models/slanet/ch_ppstructure_mobile_v2.0_SLANet_infer.tar) / [训练模型](https://paddleocr.bj.bcebos.com/ppstructure/models/slanet/ch_ppstructure_mobile_v2.0_SLANet_train.tar) |
diff --git a/ppstructure/docs/models_list_en.md b/ppstructure/docs/models_list_en.md
index e133a0bb2a..64a7cdebc3 100644
--- a/ppstructure/docs/models_list_en.md
+++ b/ppstructure/docs/models_list_en.md
@@ -35,7 +35,9 @@ If you need to use other OCR models, you can download the model in [PP-OCR model
|model| description |inference model size|download|
| --- |-----------------------------------------------------------------------------| --- | --- |
-|en_ppocr_mobile_v2.0_table_structure| Table structure model for English table scenes trained on PubTabNet dataset |18.6M|[inference model](https://paddleocr.bj.bcebos.com/dygraph_v2.0/table/en_ppocr_mobile_v2.0_table_structure_infer.tar) / [trained model](https://paddleocr.bj.bcebos.com/dygraph_v2.1/table/en_ppocr_mobile_v2.0_table_structure_train.tar) |
+|en_ppocr_mobile_v2.0_table_structure| English table recognition model trained on PubTabNet dataset based on TableRec-RARE |18.6M|[inference model](https://paddleocr.bj.bcebos.com/dygraph_v2.0/table/en_ppocr_mobile_v2.0_table_structure_infer.tar) / [trained model](https://paddleocr.bj.bcebos.com/dygraph_v2.1/table/en_ppocr_mobile_v2.0_table_structure_train.tar) |
+|en_ppstructure_mobile_v2.0_SLANet|English table recognition model trained on PubTabNet dataset based on SLANet|9M|[inference model](https://paddleocr.bj.bcebos.com/ppstructure/models/slanet/en_ppstructure_mobile_v2.0_SLANet_infer.tar) / [trained model](https://paddleocr.bj.bcebos.com/ppstructure/models/slanet/en_ppstructure_mobile_v2.0_SLANet_train.tar) |
+|ch_ppstructure_mobile_v2.0_SLANet|Chinese table recognition model trained on PubTabNet dataset based on SLANet|9.3M|[inference model](https://paddleocr.bj.bcebos.com/ppstructure/models/slanet/ch_ppstructure_mobile_v2.0_SLANet_infer.tar) / [trained model](https://paddleocr.bj.bcebos.com/ppstructure/models/slanet/ch_ppstructure_mobile_v2.0_SLANet_train.tar) |
## 3. VQA
diff --git a/ppstructure/table/README.md b/ppstructure/table/README.md
index 10308b4923..5ac99ac858 100644
--- a/ppstructure/table/README.md
+++ b/ppstructure/table/README.md
@@ -44,17 +44,24 @@ cd PaddleOCR/ppstructure
# download model
mkdir inference && cd inference
-# Download the detection model of the ultra-lightweight table English OCR model and unzip it
-wget https://paddleocr.bj.bcebos.com/dygraph_v2.0/table/en_ppocr_mobile_v2.0_table_det_infer.tar && tar xf en_ppocr_mobile_v2.0_table_det_infer.tar
-# Download the recognition model of the ultra-lightweight table English OCR model and unzip it
-wget https://paddleocr.bj.bcebos.com/dygraph_v2.0/table/en_ppocr_mobile_v2.0_table_rec_infer.tar && tar xf en_ppocr_mobile_v2.0_table_rec_infer.tar
-# Download the ultra-lightweight English table inch model and unzip it
-wget https://paddleocr.bj.bcebos.com/dygraph_v2.0/table/en_ppocr_mobile_v2.0_table_structure_infer.tar && tar xf en_ppocr_mobile_v2.0_table_structure_infer.tar
+# Download the PP-OCRv3 text detection model and unzip it
+wget https://paddleocr.bj.bcebos.com/PP-OCRv3/chinese/ch_PP-OCRv3_det_slim_infer.tar && tar xf ch_PP-OCRv3_det_slim_infer.tar
+# Download the PP-OCRv3 text recognition model and unzip it
+wget https://paddleocr.bj.bcebos.com/PP-OCRv3/chinese/ch_PP-OCRv3_rec_slim_infer.tar && tar xf ch_PP-OCRv3_rec_slim_infer.tar
+# Download the PP-Structurev2 form recognition model and unzip it
+wget https://paddleocr.bj.bcebos.com/ppstructure/models/slanet/ch_ppstructure_mobile_v2.0_SLANet_infer.tar && tar xf ch_ppstructure_mobile_v2.0_SLANet_infer.tar
cd ..
# run
-python3 table/predict_table.py --det_model_dir=inference/en_ppocr_mobile_v2.0_table_det_infer --rec_model_dir=inference/en_ppocr_mobile_v2.0_table_rec_infer --table_model_dir=inference/en_ppocr_mobile_v2.0_table_structure_infer --image_dir=./docs/table/table.jpg --rec_char_dict_path=../ppocr/utils/dict/table_dict.txt --table_char_dict_path=../ppocr/utils/dict/table_structure_dict.txt --det_limit_side_len=736 --det_limit_type=min --output ./output/table
+python3.7 table/predict_table.py \
+ --det_model_dir=inference/ch_PP-OCRv3_det_slim_infer \
+ --rec_model_dir=inference/ch_PP-OCRv3_rec_slim_infer \
+ --table_model_dir=inference/ch_ppstructure_mobile_v2.0_SLANet_infer \
+ --rec_char_dict_path=../ppocr/utils/ppocr_keys_v1.txt \
+ --table_char_dict_path=../ppocr/utils/dict/table_structure_dict_ch.txt \
+ --image_dir=docs/table/table.jpg \
+ --output=../output/table
+
```
-Note: The above model is trained on the PubLayNet dataset and only supports English scanning scenarios. If you need to identify other scenarios, you need to train the model yourself and replace the three fields `det_model_dir`, `rec_model_dir`, `table_model_dir`.
After the operation is completed, the excel table of each image will be saved to the directory specified by the output field, and an html file will be produced in the directory to visually view the cell coordinates and the recognized table.
diff --git a/ppstructure/table/README_ch.md b/ppstructure/table/README_ch.md
index 3f31c0106b..a16de938a6 100644
--- a/ppstructure/table/README_ch.md
+++ b/ppstructure/table/README_ch.md
@@ -54,20 +54,25 @@ cd PaddleOCR/ppstructure
# 下载模型
mkdir inference && cd inference
-# 下载超轻量级表格英文OCR模型的检测模型并解压
-wget https://paddleocr.bj.bcebos.com/dygraph_v2.0/table/en_ppocr_mobile_v2.0_table_det_infer.tar && tar xf en_ppocr_mobile_v2.0_table_det_infer.tar
-# 下载超轻量级表格英文OCR模型的识别模型并解压
-wget https://paddleocr.bj.bcebos.com/dygraph_v2.0/table/en_ppocr_mobile_v2.0_table_rec_infer.tar && tar xf en_ppocr_mobile_v2.0_table_rec_infer.tar
-# 下载超轻量级英文表格英寸模型并解压
-wget https://paddleocr.bj.bcebos.com/dygraph_v2.0/table/en_ppocr_mobile_v2.0_table_structure_infer.tar && tar xf en_ppocr_mobile_v2.0_table_structure_infer.tar
+# 下载PP-OCRv3文本检测模型并解压
+wget https://paddleocr.bj.bcebos.com/PP-OCRv3/chinese/ch_PP-OCRv3_det_slim_infer.tar && tar xf ch_PP-OCRv3_det_slim_infer.tar
+# 下载PP-OCRv3文本识别模型并解压
+wget https://paddleocr.bj.bcebos.com/PP-OCRv3/chinese/ch_PP-OCRv3_rec_slim_infer.tar && tar xf ch_PP-OCRv3_rec_slim_infer.tar
+# 下载PP-Structurev2表格识别模型并解压
+wget https://paddleocr.bj.bcebos.com/ppstructure/models/slanet/ch_ppstructure_mobile_v2.0_SLANet_infer.tar && tar xf ch_ppstructure_mobile_v2.0_SLANet_infer.tar
cd ..
# 执行预测
-python3 table/predict_table.py --det_model_dir=inference/en_ppocr_mobile_v2.0_table_det_infer --rec_model_dir=inference/en_ppocr_mobile_v2.0_table_rec_infer --table_model_dir=inference/en_ppocr_mobile_v2.0_table_structure_infer --image_dir=./docs/table/table.jpg --rec_char_dict_path=../ppocr/utils/dict/table_dict.txt --table_char_dict_path=../ppocr/utils/dict/table_structure_dict.txt --det_limit_side_len=736 --det_limit_type=min --output ./output/table
+python3.7 table/predict_table.py \
+ --det_model_dir=inference/ch_PP-OCRv3_det_slim_infer \
+ --rec_model_dir=inference/ch_PP-OCRv3_rec_slim_infer \
+ --table_model_dir=inference/ch_ppstructure_mobile_v2.0_SLANet_infer \
+ --rec_char_dict_path=../ppocr/utils/ppocr_keys_v1.txt \
+ --table_char_dict_path=../ppocr/utils/dict/table_structure_dict_ch.txt \
+ --image_dir=docs/table/table.jpg \
+ --output=../output/table
```
运行完成后,每张图片的excel表格会保存到output字段指定的目录下,同时在该目录下回生产一个html文件,用于可视化查看单元格坐标和识别的表格。
-note: 上述模型是在 PubLayNet 数据集上训练的表格识别模型,仅支持英文扫描场景,如需识别其他场景需要自己训练模型后替换 `det_model_dir`,`rec_model_dir`,`table_model_dir`三个字段即可。
-
### 3.2 训练
diff --git a/ppstructure/utility.py b/ppstructure/utility.py
index 3e5054a7d3..cda4c063bc 100644
--- a/ppstructure/utility.py
+++ b/ppstructure/utility.py
@@ -28,7 +28,7 @@ def init_args():
parser.add_argument("--table_algorithm", type=str, default='TableAttn')
parser.add_argument("--table_model_dir", type=str)
parser.add_argument(
- "--merge_no_span_structure", type=str2bool, default=False)
+ "--merge_no_span_structure", type=str2bool, default=True)
parser.add_argument(
"--table_char_dict_path",
type=str,
From b26ce23774a06a550c472867e49138b7758c1823 Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Tue, 16 Aug 2022 10:55:24 +0000
Subject: [PATCH 33/49] rm unused code
---
paddleocr.py | 27 +++++++++------------------
1 file changed, 9 insertions(+), 18 deletions(-)
diff --git a/paddleocr.py b/paddleocr.py
index fb1427b83f..b5bb5d21c3 100644
--- a/paddleocr.py
+++ b/paddleocr.py
@@ -318,10 +318,10 @@ def parse_args(mMain=True):
"--structure_version",
type=str,
choices=SUPPORT_STRUCTURE_MODEL_VERSION,
- default='PP-Structure',
+ default='PP-Structurev2',
help='Model version, the current model support list is as follows:'
' 1. PP-Structure Support en table structure model.'
- ' 2. PP-Structure Support ch and en table structure model.')
+ ' 2. PP-Structurev2 Support ch and en table structure model.')
for action in parser._actions:
if action.dest in [
@@ -529,6 +529,12 @@ class PPStructure(StructureSystem):
if not params.show_log:
logger.setLevel(logging.INFO)
lang, det_lang = parse_lang(params.lang)
+ if lang == 'ch':
+ table_lang = 'ch'
+ else:
+ table_lang = 'en'
+ if params.structure_version == 'PP-Structure':
+ params.merge_no_span_structure = False
# init model dir
det_model_config = get_model_config('OCR', params.ocr_version, 'det',
@@ -543,7 +549,7 @@ class PPStructure(StructureSystem):
params.rec_model_dir,
os.path.join(BASE_DIR, 'whl', 'rec', lang), rec_model_config['url'])
table_model_config = get_model_config(
- 'STRUCTURE', params.structure_version, 'table', 'ch')
+ 'STRUCTURE', params.structure_version, 'table', table_lang)
params.table_model_dir, table_url = confirm_model_dir_url(
params.table_model_dir,
os.path.join(BASE_DIR, 'whl', 'table'), table_model_config['url'])
@@ -629,18 +635,3 @@ def main():
for item in result:
item.pop('img')
logger.info(item)
-
-
-if __name__ == "__main__":
- table_engine = PPStructure(layout=False, show_log=True)
-
- save_folder = './output'
- img_path = 'ppstructure/docs/table/table.jpg'
- img = cv2.imread(img_path)
- result = table_engine(img)
- save_structure_res(result, save_folder,
- os.path.basename(img_path).split('.')[0])
-
- for line in result:
- line.pop('img')
- print(line)
From cf6f7012ad20e3cc16e5dac1ad264177d57d56cb Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Tue, 16 Aug 2022 12:42:47 +0000
Subject: [PATCH 34/49] add structure predict doc
---
ppstructure/table/README.md | 20 ++++++++++++++++++++
ppstructure/table/README_ch.md | 21 ++++++++++++++++++++-
ppstructure/table/predict_structure.py | 5 ++++-
3 files changed, 44 insertions(+), 2 deletions(-)
diff --git a/ppstructure/table/README.md b/ppstructure/table/README.md
index 5ac99ac858..7ecbe0ad84 100644
--- a/ppstructure/table/README.md
+++ b/ppstructure/table/README.md
@@ -39,6 +39,8 @@ We evaluated the algorithm on the PubTabNet[1] eval dataset, and the
### 3.1 quick start
+- table recognition
+
```python
cd PaddleOCR/ppstructure
@@ -65,6 +67,24 @@ python3.7 table/predict_table.py \
After the operation is completed, the excel table of each image will be saved to the directory specified by the output field, and an html file will be produced in the directory to visually view the cell coordinates and the recognized table.
+- table structure recognition
+```python
+cd PaddleOCR/ppstructure
+
+# download model
+mkdir inference && cd inference
+# Download the PP-Structurev2 form recognition model and unzip it
+wget https://paddleocr.bj.bcebos.com/ppstructure/models/slanet/ch_ppstructure_mobile_v2.0_SLANet_infer.tar && tar xf ch_ppstructure_mobile_v2.0_SLANet_infer.tar
+cd ..
+# run
+python3.7 table/predict_structure.py \
+ --table_model_dir=inference/ch_ppstructure_mobile_v2.0_SLANet_infer \
+ --table_char_dict_path=../ppocr/utils/dict/table_structure_dict_ch.txt \
+ --image_dir=docs/table/table.jpg \
+ --output=../output/table
+```
+After the run is complete, the visualization of the detection frame of the cell will be saved to the directory specified by the output field.
+
### 3.2 Train
In this chapter, we only introduce the training of the table structure model, For model training of [text detection](../../doc/doc_en/detection_en.md) and [text recognition](../../doc/doc_en/recognition_en.md), please refer to the corresponding documents
diff --git a/ppstructure/table/README_ch.md b/ppstructure/table/README_ch.md
index a16de938a6..ac5029e7ff 100644
--- a/ppstructure/table/README_ch.md
+++ b/ppstructure/table/README_ch.md
@@ -49,6 +49,7 @@
### 3.1 快速开始
+- 表格识别
```python
cd PaddleOCR/ppstructure
@@ -61,7 +62,7 @@ wget https://paddleocr.bj.bcebos.com/PP-OCRv3/chinese/ch_PP-OCRv3_rec_slim_infer
# 下载PP-Structurev2表格识别模型并解压
wget https://paddleocr.bj.bcebos.com/ppstructure/models/slanet/ch_ppstructure_mobile_v2.0_SLANet_infer.tar && tar xf ch_ppstructure_mobile_v2.0_SLANet_infer.tar
cd ..
-# 执行预测
+# 执行表格识别
python3.7 table/predict_table.py \
--det_model_dir=inference/ch_PP-OCRv3_det_slim_infer \
--rec_model_dir=inference/ch_PP-OCRv3_rec_slim_infer \
@@ -73,6 +74,24 @@ python3.7 table/predict_table.py \
```
运行完成后,每张图片的excel表格会保存到output字段指定的目录下,同时在该目录下回生产一个html文件,用于可视化查看单元格坐标和识别的表格。
+- 表格结构识别
+```python
+cd PaddleOCR/ppstructure
+
+# 下载模型
+mkdir inference && cd inference
+# 下载PP-Structurev2表格识别模型并解压
+wget https://paddleocr.bj.bcebos.com/ppstructure/models/slanet/ch_ppstructure_mobile_v2.0_SLANet_infer.tar && tar xf ch_ppstructure_mobile_v2.0_SLANet_infer.tar
+cd ..
+# 执行表格结构识别
+python3.7 table/predict_structure.py \
+ --table_model_dir=inference/ch_ppstructure_mobile_v2.0_SLANet_infer \
+ --table_char_dict_path=../ppocr/utils/dict/table_structure_dict_ch.txt \
+ --image_dir=docs/table/table.jpg \
+ --output=../output/table
+```
+运行完成后,单元格的检测框可视化会保存到output字段指定的目录下。
+
### 3.2 训练
diff --git a/ppstructure/table/predict_structure.py b/ppstructure/table/predict_structure.py
index c4a816fd87..7198fb2bcd 100755
--- a/ppstructure/table/predict_structure.py
+++ b/ppstructure/table/predict_structure.py
@@ -147,7 +147,10 @@ def main(args):
f_w.write("result: {}, {}\n".format(structure_str_list,
bbox_list_str))
- img = draw_rectangle(image_file, bbox_list)
+ if len(bbox_list) > 0 and len(bbox_list[0]) == 4:
+ img = draw_rectangle(image_file, pred_res['cell_bbox'])
+ else:
+ img = utility.draw_boxes(img, bbox_list)
img_save_path = os.path.join(args.output,
os.path.basename(image_file))
cv2.imwrite(img_save_path, img)
From 80af73ca5add5491ea03dfb2de148b979c4a32b4 Mon Sep 17 00:00:00 2001
From: WenmuZhou <572459439@qq.com>
Date: Tue, 16 Aug 2022 14:16:24 +0000
Subject: [PATCH 35/49] update doc
---
doc/doc_ch/table_recognition.md | 310 +++++++++++++++++++++++
doc/doc_en/table_recognition_en.md | 320 ++++++++++++++++++++++++
ppstructure/docs/imgs/slanet_result.jpg | Bin 0 -> 79603 bytes
ppstructure/table/README.md | 114 +++------
ppstructure/table/README_ch.md | 116 +++------
5 files changed, 695 insertions(+), 165 deletions(-)
create mode 100644 doc/doc_ch/table_recognition.md
create mode 100644 doc/doc_en/table_recognition_en.md
create mode 100644 ppstructure/docs/imgs/slanet_result.jpg
diff --git a/doc/doc_ch/table_recognition.md b/doc/doc_ch/table_recognition.md
new file mode 100644
index 0000000000..fea95222cf
--- /dev/null
+++ b/doc/doc_ch/table_recognition.md
@@ -0,0 +1,310 @@
+# 表格识别
+
+本文提供了PaddleOCR表格识别模型的全流程指南,包括数据准备、模型训练、调优、评估、预测,各个阶段的详细说明:
+
+- [1. 数据准备](#1-数据准备)
+ - [1.1. 准备数据集](#11-准备数据集)
+ - [1.2. 数据下载](#12-数据下载)
+ - [1.3. 数据集生成](#13-数据集生成)
+- [2. 开始训练](#2-开始训练)
+ - [2.1. 启动训练](#21-启动训练)
+ - [2.2. 断点训练](#22-断点训练)
+ - [2.3. 更换Backbone 训练](#23-更换backbone-训练)
+ - [2.4. 混合精度训练](#24-混合精度训练)
+ - [2.5. 分布式训练](#25-分布式训练)
+ - [2.6. 知识蒸馏训练](#26-知识蒸馏训练)
+ - [2.7. 其他训练环境](#27-其他训练环境)
+ - [2.8 模型微调](#28-模型微调)
+- [3. 模型评估与预测](#3-模型评估与预测)
+ - [3.1. 指标评估](#31-指标评估)
+ - [3.2. 测试表格结构识别效果](#32-测试表格结构识别效果)
+- [4. 模型导出与预测](#4-模型导出与预测)
+- [5. FAQ](#5-faq)
+
+# 1. 数据准备
+
+## 1.1. 准备数据集
+
+PaddleOCR 表格识别模型数据集格式如下:
+```txt
+img_label # 每张图片标注经过json.dumps()之后的字符串
+...
+img_label
+```
+
+每一行的json格式为:
+```json
+{
+ 'filename': PMC5755158_010_01.png, # 图像名
+ 'split': ’train‘, # 图像属于训练集还是验证集
+ 'imgid': 0, # 图像的index
+ 'html': {
+ 'structure': {'tokens': ['', '', '', ...]}, # 表格的HTML字符串
+ 'cell': [
+ {
+ 'tokens': ['P', 'a', 'd', 'd', 'l', 'e', 'P', 'a', 'd', 'd', 'l', 'e'], # 表格中的单个文本
+ 'bbox': [x0, y0, x1, y1] # 表格中的单个文本的坐标
+ }
+ ]
+ }
+}
+```
+
+训练数据的默认存储路径是 `PaddleOCR/train_data`,如果您的磁盘上已有数据集,只需创建软链接至数据集目录:
+
+```
+# linux and mac os
+ln -sf /train_data/dataset
+# windows
+mklink /d /train_data/dataset
+```
+
+## 1.2. 数据下载
+
+公开数据集下载可参考 [table_datasets](dataset/table_datasets.md)。
+
+## 1.3. 数据集生成
+
+使用[TableGeneration](https://github.com/WenmuZhou/TableGeneration)可进行扫描表格图像的生成。
+
+TableGeneration是一个开源表格数据集生成工具,其通过浏览器渲染的方式对html字符串进行渲染后获得表格图像。部分样张如下:
+
+|类型|样例|
+|---|---|
+|简单表格||
+|彩色表格||
+
+# 2. 开始训练
+
+PaddleOCR提供了训练脚本、评估脚本和预测脚本,本节将以 [SLANet](../../configs/table/SLANet.yml) 模型训练PubTabNet英文数据集为例:
+
+## 2.1. 启动训练
+
+*如果您安装的是cpu版本,请将配置文件中的 `use_gpu` 字段修改为false*
+
+```
+# GPU训练 支持单卡,多卡训练
+# 训练日志会自动保存为 "{save_model_dir}" 下的train.log
+
+#单卡训练(训练周期长,不建议)
+python3 tools/train.py -c configs/table/SLANet.yml
+
+#多卡训练,通过--gpus参数指定卡号
+python3 -m paddle.distributed.launch --gpus '0,1,2,3' tools/train.py -c configs/table/SLANet.yml
+```
+
+正常启动训练后,会看到以下log输出:
+
+```
+[2022/08/16 03:07:33] ppocr INFO: epoch: [1/400], global_step: 20, lr: 0.000100, acc: 0.000000, loss: 3.915012, structure_loss: 3.229450, loc_loss: 0.670590, avg_reader_cost: 2.63382 s, avg_batch_cost: 6.32390 s, avg_samples: 48.0, ips: 7.59025 samples/s, eta: 9 days, 2:29:27
+[2022/08/16 03:08:41] ppocr INFO: epoch: [1/400], global_step: 40, lr: 0.000100, acc: 0.000000, loss: 1.750859, structure_loss: 1.082116, loc_loss: 0.652822, avg_reader_cost: 0.02533 s, avg_batch_cost: 3.37251 s, avg_samples: 48.0, ips: 14.23271 samples/s, eta: 6 days, 23:28:43
+[2022/08/16 03:09:46] ppocr INFO: epoch: [1/400], global_step: 60, lr: 0.000100, acc: 0.000000, loss: 1.395154, structure_loss: 0.776803, loc_loss: 0.625030, avg_reader_cost: 0.02550 s, avg_batch_cost: 3.26261 s, avg_samples: 48.0, ips: 14.71214 samples/s, eta: 6 days, 5:11:48
+```
+
+log 中自动打印如下信息:
+
+| 字段 | 含义 |
+| :----: | :------: |
+| epoch | 当前迭代轮次 |
+| global_step | 当前迭代次数 |
+| lr | 当前学习率 |
+| acc | 当前batch的准确率 |
+| loss | 当前损失函数 |
+| structure_loss | 表格结构损失值 |
+| loc_loss | 单元格坐标损失值 |
+| avg_reader_cost | 当前 batch 数据处理耗时 |
+| avg_batch_cost | 当前 batch 总耗时 |
+| avg_samples | 当前 batch 内的样本数 |
+| ips | 每秒处理图片的数量 |
+
+
+PaddleOCR支持训练和评估交替进行, 可以在 `configs/table/SLANet.yml` 中修改 `eval_batch_step` 设置评估频率,默认每1000个iter评估一次。评估过程中默认将最佳acc模型,保存为 `output/SLANet/best_accuracy` 。
+
+如果验证集很大,测试将会比较耗时,建议减少评估次数,或训练完再进行评估。
+
+**提示:** 可通过 -c 参数选择 `configs/table/` 路径下的多种模型配置进行训练,PaddleOCR支持的表格识别算法可以参考[前沿算法列表](https://github.com/PaddlePaddle/PaddleOCR/blob/dygraph/doc/doc_ch/algorithm_overview.md#3-%E8%A1%A8%E6%A0%BC%E8%AF%86%E5%88%AB%E7%AE%97%E6%B3%95):
+
+**注意,预测/评估时的配置文件请务必与训练一致。**
+
+## 2.2. 断点训练
+
+如果训练程序中断,如果希望加载训练中断的模型从而恢复训练,可以通过指定Global.checkpoints指定要加载的模型路径:
+```shell
+python3 tools/train.py -c configs/table/SLANet.yml -o Global.checkpoints=./your/trained/model
+```
+
+**注意**:`Global.checkpoints`的优先级高于`Global.pretrained_model`的优先级,即同时指定两个参数时,优先加载`Global.checkpoints`指定的模型,如果`Global.checkpoints`指定的模型路径有误,会加载`Global.pretrained_model`指定的模型。
+
+## 2.3. 更换Backbone 训练
+
+PaddleOCR将网络划分为四部分,分别在[ppocr/modeling](../../ppocr/modeling)下。 进入网络的数据将按照顺序(transforms->backbones->necks->heads)依次通过这四个部分。
+
+```bash
+├── architectures # 网络的组网代码
+├── transforms # 网络的图像变换模块
+├── backbones # 网络的特征提取模块
+├── necks # 网络的特征增强模块
+└── heads # 网络的输出模块
+```
+如果要更换的Backbone 在PaddleOCR中有对应实现,直接修改配置yml文件中`Backbone`部分的参数即可。
+
+如果要使用新的Backbone,更换backbones的例子如下:
+
+1. 在 [ppocr/modeling/backbones](../../ppocr/modeling/backbones) 文件夹下新建文件,如my_backbone.py。
+2. 在 my_backbone.py 文件内添加相关代码,示例代码如下:
+
+```python
+import paddle
+import paddle.nn as nn
+import paddle.nn.functional as F
+
+
+class MyBackbone(nn.Layer):
+ def __init__(self, *args, **kwargs):
+ super(MyBackbone, self).__init__()
+ # your init code
+ self.conv = nn.xxxx
+
+ def forward(self, inputs):
+ # your network forward
+ y = self.conv(inputs)
+ return y
+```
+
+3. 在 [ppocr/modeling/backbones/\__init\__.py](../../ppocr/modeling/backbones/__init__.py)文件内导入添加的`MyBackbone`模块,然后修改配置文件中Backbone进行配置即可使用,格式如下:
+
+```yaml
+Backbone:
+name: MyBackbone
+args1: args1
+```
+
+**注意**:如果要更换网络的其他模块,可以参考[文档](./add_new_algorithm.md)。
+
+## 2.4. 混合精度训练
+
+如果您想进一步加快训练速度,可以使用[自动混合精度训练](https://www.paddlepaddle.org.cn/documentation/docs/zh/guides/01_paddle2.0_introduction/basic_concept/amp_cn.html), 以单机单卡为例,命令如下:
+
+```shell
+python3 tools/train.py -c configs/table/SLANet.yml \
+ -o Global.pretrained_model=./pretrain_models/SLANet/best_accuracy \
+ Global.use_amp=True Global.scale_loss=1024.0 Global.use_dynamic_loss_scaling=True
+ ```
+
+## 2.5. 分布式训练
+
+多机多卡训练时,通过 `--ips` 参数设置使用的机器IP地址,通过 `--gpus` 参数设置使用的GPU ID:
+
+```bash
+python3 -m paddle.distributed.launch --ips="xx.xx.xx.xx,xx.xx.xx.xx" --gpus '0,1,2,3' tools/train.py -c configs/table/SLANet.yml \
+ -o Global.pretrained_model=./pretrain_models/SLANet/best_accuracy
+```
+
+**注意:** (1)采用多机多卡训练时,需要替换上面命令中的ips值为您机器的地址,机器之间需要能够相互ping通;(2)训练时需要在多个机器上分别启动命令。查看机器ip地址的命令为`ifconfig`;(3)更多关于分布式训练的性能优势等信息,请参考:[分布式训练教程](./distributed_training.md)。
+
+## 2.6. 知识蒸馏训练
+
+coming soon!
+
+## 2.7. 其他训练环境
+
+- Windows GPU/CPU
+在Windows平台上与Linux平台略有不同:
+Windows平台只支持`单卡`的训练与预测,指定GPU进行训练`set CUDA_VISIBLE_DEVICES=0`
+在Windows平台,DataLoader只支持单进程模式,因此需要设置 `num_workers` 为0;
+
+- macOS
+不支持GPU模式,需要在配置文件中设置`use_gpu`为False,其余训练评估预测命令与Linux GPU完全相同。
+
+- Linux DCU
+DCU设备上运行需要设置环境变量 `export HIP_VISIBLE_DEVICES=0,1,2,3`,其余训练评估预测命令与Linux GPU完全相同。
+
+## 2.8 模型微调
+
+实际使用过程中,建议加载官方提供的预训练模型,在自己的数据集中进行微调,关于模型的微调方法,请参考:[模型微调教程](./finetune.md)。
+
+
+# 3. 模型评估与预测
+
+## 3.1. 指标评估
+
+训练中模型参数默认保存在`Global.save_model_dir`目录下。在评估指标时,需要设置`Global.checkpoints`指向保存的参数文件。评估数据集可以通过 `configs/table/SLANet.yml` 修改Eval中的 `label_file_list` 设置。
+
+
+```
+# GPU 评估, Global.checkpoints 为待测权重
+python3 -m paddle.distributed.launch --gpus '0' tools/eval.py -c configs/table/SLANet.yml -o Global.checkpoints={path/to/weights}/best_accuracy
+```
+
+## 3.2. 测试表格结构识别效果
+
+使用 PaddleOCR 训练好的模型,可以通过以下脚本进行快速预测。
+
+默认预测图片存储在 `infer_img` 里,通过 `-o Global.checkpoints` 加载训练好的参数文件:
+
+根据配置文件中设置的 `save_model_dir` 和 `save_epoch_step` 字段,会有以下几种参数被保存下来:
+
+```
+output/SLANet/
+├── best_accuracy.pdopt
+├── best_accuracy.pdparams
+├── best_accuracy.states
+├── config.yml
+├── latest.pdopt
+├── latest.pdparams
+├── latest.states
+└── train.log
+```
+其中 best_accuracy.* 是评估集上的最优模型;latest.* 是最后一个epoch的模型。
+
+```
+# 预测表格图像
+python3 tools/infer_table.py -c configs/table/SLANet.yml -o Global.pretrained_model={path/to/weights}/best_accuracy Global.infer_img=ppstructure/docs/table/table.jpg
+```
+
+预测图片:
+
+
+
+得到输入图像的预测结果:
+
+```
+['', '', '', '', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', '', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' ', '', ''],[[320.0562438964844, 197.83375549316406, 350.0928955078125, 214.4309539794922], ... , [318.959228515625, 271.0166931152344, 353.7394104003906, 286.4538269042969]]
+```
+
+单元格坐标可视化结果为
+
+
+
+# 4. 模型导出与预测
+
+inference 模型(`paddle.jit.save`保存的模型)
+一般是模型训练,把模型结构和模型参数保存在文件中的固化模型,多用于预测部署场景。
+训练过程中保存的模型是checkpoints模型,保存的只有模型的参数,多用于恢复训练等。
+与checkpoints模型相比,inference 模型会额外保存模型的结构信息,在预测部署、加速推理上性能优越,灵活方便,适合于实际系统集成。
+
+表格识别模型转inference模型与文字检测识别的方式相同,如下:
+
+```
+# -c 后面设置训练算法的yml配置文件
+# -o 配置可选参数
+# Global.pretrained_model 参数设置待转换的训练模型地址,不用添加文件后缀 .pdmodel,.pdopt或.pdparams。
+# Global.save_inference_dir参数设置转换的模型将保存的地址。
+
+python3 tools/export_model.py -c configs/table/SLANet.yml -o Global.pretrained_model=./pretrain_models/SLANet/best_accuracy Global.save_inference_dir=./inference/SLANet/
+```
+
+转换成功后,在目录下有三个文件:
+
+```
+inference/SLANet/
+ ├── inference.pdiparams # inference模型的参数文件
+ ├── inference.pdiparams.info # inference模型的参数信息,可忽略
+ └── inference.pdmodel # inference模型的program文件
+```
+
+# 5. FAQ
+
+Q1: 训练模型转inference 模型之后预测效果不一致?
+
+**A**:此类问题出现较多,问题多是trained model预测时候的预处理、后处理参数和inference model预测的时候的预处理、后处理参数不一致导致的。可以对比训练使用的配置文件中的预处理、后处理和预测时是否存在差异。
diff --git a/doc/doc_en/table_recognition_en.md b/doc/doc_en/table_recognition_en.md
new file mode 100644
index 0000000000..28f8c6fa98
--- /dev/null
+++ b/doc/doc_en/table_recognition_en.md
@@ -0,0 +1,320 @@
+# Table Recognition
+
+This article provides a full-process guide for the PaddleOCR table recognition model, including data preparation, model training, tuning, evaluation, prediction, and detailed descriptions of each stage:
+
+- [1. Data Preparation](#1-data-preparation)
+ - [1.1. DataSet Preparation](#11-dataset-preparation)
+ - [1.2. Data Download](#12-data-download)
+ - [1.3. Dataset Generation](#13-dataset-generation)
+- [2. Training](#2-training)
+ - [2.1. Start Training](#21-start-training)
+ - [2.2. Resume Training](#22-resume-training)
+ - [2.3. Training with New Backbone](#23-training-with-new-backbone)
+ - [2.4. Mixed Precision Training](#24-mixed-precision-training)
+ - [2.5. Distributed Training](#25-distributed-training)
+ - [2.6. Training with Knowledge Distillation](#26-training-with-knowledge-distillation)
+ - [2.7. Training on other platform(Windows/macOS/Linux DCU)](#27-training-on-other-platformwindowsmacoslinux-dcu)
+ - [2.8 Fine-tuning](#28-fine-tuning)
+- [3. Evaluation and Test](#3-evaluation-and-test)
+ - [3.1. Evaluation](#31-evaluation)
+ - [3.2. Test table structure recognition effect](#32-test-table-structure-recognition-effect)
+- [4. Model export and prediction](#4-model-export-and-prediction)
+ - [5. FAQ](#5-faq)
+
+# 1. Data Preparation
+
+## 1.1. DataSet Preparation
+
+The format of the PaddleOCR table recognition model dataset is as follows:
+```txt
+img_label # Each image is marked with a string after json.dumps()
+...
+img_label
+```
+
+The json format of each line is:
+```json
+{
+ 'filename': PMC5755158_010_01.png, # image name
+ 'split': ’train‘, # whether the image belongs to the training set or the validation set
+ 'imgid': 0, # index of image
+ 'html': {
+ 'structure': {'tokens': ['', '', '', ...]}, # HTML string of the table
+ 'cell': [
+ {
+ 'tokens': ['P', 'a', 'd', 'd', 'l', 'e', 'P', 'a', 'd', 'd', 'l', 'e'], # text in cell
+ 'bbox': [x0, y0, x1, y1] # bbox of cell
+ }
+ ]
+ }
+}
+```
+
+The default storage path for training data is `PaddleOCR/train_data`, if you already have a dataset on disk, just create a soft link to the dataset directory:
+
+```
+# linux and mac os
+ln -sf /train_data/dataset
+# windows
+mklink /d /train_data/dataset
+```
+
+## 1.2. Data Download
+
+Download the public dataset reference [table_datasets](dataset/table_datasets_en.md)。
+
+## 1.3. Dataset Generation
+
+Use [TableGeneration](https://github.com/WenmuZhou/TableGeneration) to generate scanned table images.
+
+TableGeneration is an open source table dataset generation tool, which renders html strings through browser rendering to obtain table images.
+
+Some samples are as follows:
+
+|Type|Sample|
+|---|---|
+|Simple Table||
+|Simple Color Table||
+
+# 2. Training
+
+PaddleOCR provides training scripts, evaluation scripts, and prediction scripts. In this section, the [SLANet](../../configs/table/SLANet.yml) model will be used as an example:
+
+## 2.1. Start Training
+
+*If you are installing the cpu version, please modify the `use_gpu` field in the configuration file to false*
+
+```
+# GPU training Support single card and multi-card training
+# The training log will be automatically saved as train.log under "{save_model_dir}"
+
+# specify the single card training(Long training time, not recommended)
+python3 tools/train.py -c configs/table/SLANet.yml
+
+# specify the card number through --gpus
+python3 -m paddle.distributed.launch --gpus '0,1,2,3' tools/train.py -c configs/table/SLANet.yml
+```
+
+After starting training normally, you will see the following log output:
+
+```
+[2022/08/16 03:07:33] ppocr INFO: epoch: [1/400], global_step: 20, lr: 0.000100, acc: 0.000000, loss: 3.915012, structure_loss: 3.229450, loc_loss: 0.670590, avg_reader_cost: 2.63382 s, avg_batch_cost: 6.32390 s, avg_samples: 48.0, ips: 7.59025 samples/s, eta: 9 days, 2:29:27
+[2022/08/16 03:08:41] ppocr INFO: epoch: [1/400], global_step: 40, lr: 0.000100, acc: 0.000000, loss: 1.750859, structure_loss: 1.082116, loc_loss: 0.652822, avg_reader_cost: 0.02533 s, avg_batch_cost: 3.37251 s, avg_samples: 48.0, ips: 14.23271 samples/s, eta: 6 days, 23:28:43
+[2022/08/16 03:09:46] ppocr INFO: epoch: [1/400], global_step: 60, lr: 0.000100, acc: 0.000000, loss: 1.395154, structure_loss: 0.776803, loc_loss: 0.625030, avg_reader_cost: 0.02550 s, avg_batch_cost: 3.26261 s, avg_samples: 48.0, ips: 14.71214 samples/s, eta: 6 days, 5:11:48
+```
+
+The following information is automatically printed in the log:
+
+| Field | Meaning |
+| :----: | :------: |
+| epoch | current iteration round |
+| global_step | current iteration count |
+| lr | current learning rate |
+| acc | The accuracy of the current batch |
+| loss | current loss function |
+| structure_loss | Table Structure Loss Values |
+| loc_loss | Cell Coordinate Loss Value |
+| avg_reader_cost | Current batch data processing time |
+| avg_batch_cost | The total time spent in the current batch |
+| avg_samples | The number of samples in the current batch |
+| ips | Number of images processed per second |
+
+
+PaddleOCR supports alternating training and evaluation. You can modify `eval_batch_step` in `configs/table/SLANet.yml` to set the evaluation frequency. By default, it is evaluated once every 1000 iters. During the evaluation process, the best acc model is saved as `output/SLANet/best_accuracy` by default.
+
+If the validation set is large, the test will be time-consuming. It is recommended to reduce the number of evaluations, or perform evaluation after training.
+
+**Tips:** You can use the -c parameter to select various model configurations under the `configs/table/` path for training. For the table recognition algorithms supported by PaddleOCR, please refer to [Table Algorithms List](https://github.com/PaddlePaddle/PaddleOCR/blob/dygraph/doc/doc_en/algorithm_overview_en.md#3):
+
+**Note that the configuration file for prediction/evaluation must be the same as training. **
+
+## 2.2. Resume Training
+
+If the training program is interrupted, if you want to load the interrupted model to resume training, you can specify the path of the model to be loaded by specifying Global.checkpoints:
+
+```shell
+python3 tools/train.py -c configs/table/SLANet.yml -o Global.checkpoints=./your/trained/model
+```
+**Note**: The priority of `Global.checkpoints` is higher than that of `Global.pretrained_model`, that is, when two parameters are specified at the same time, the model specified by `Global.checkpoints` will be loaded first. If `Global.checkpoints` The specified model path is incorrect, and the model specified by `Global.pretrained_model` will be loaded.
+
+## 2.3. Training with New Backbone
+
+The network part completes the construction of the network, and PaddleOCR divides the network into four parts, which are under [ppocr/modeling](../../ppocr/modeling). The data entering the network will pass through these four parts in sequence(transforms->backbones->
+necks->heads).
+
+```bash
+├── architectures # Code for building network
+├── transforms # Image Transformation Module
+├── backbones # Feature extraction module
+├── necks # Feature enhancement module
+└── heads # Output module
+```
+
+If the Backbone to be replaced has a corresponding implementation in PaddleOCR, you can directly modify the parameters in the `Backbone` part of the configuration yml file.
+
+However, if you want to use a new Backbone, an example of replacing the backbones is as follows:
+
+1. Create a new file under the [ppocr/modeling/backbones](../../ppocr/modeling/backbones) folder, such as my_backbone.py.
+2. Add code in the my_backbone.py file, the sample code is as follows:
+
+```python
+import paddle
+import paddle.nn as nn
+import paddle.nn.functional as F
+
+
+class MyBackbone(nn.Layer):
+ def __init__(self, *args, **kwargs):
+ super(MyBackbone, self).__init__()
+ # your init code
+ self.conv = nn.xxxx
+
+ def forward(self, inputs):
+ # your network forward
+ y = self.conv(inputs)
+ return y
+```
+
+3. Import the added module in the [ppocr/modeling/backbones/\__init\__.py](../../ppocr/modeling/backbones/__init__.py) file.
+
+After adding the four-part modules of the network, you only need to configure them in the configuration file to use, such as:
+
+```yaml
+ Backbone:
+ name: MyBackbone
+ args1: args1
+```
+
+**NOTE**: More details about replace Backbone and other mudule can be found in [doc](add_new_algorithm_en.md).
+
+## 2.4. Mixed Precision Training
+
+If you want to speed up your training further, you can use [Auto Mixed Precision Training](https://www.paddlepaddle.org.cn/documentation/docs/zh/guides/01_paddle2.0_introduction/basic_concept/amp_cn.html), taking a single machine and a single gpu as an example, the commands are as follows:
+
+```shell
+python3 tools/train.py -c configs/table/SLANet.yml \
+ -o Global.pretrained_model=./pretrain_models/SLANet/best_accuracy \
+ Global.use_amp=True Global.scale_loss=1024.0 Global.use_dynamic_loss_scaling=True
+ ```
+
+## 2.5. Distributed Training
+
+During multi-machine multi-gpu training, use the `--ips` parameter to set the used machine IP address, and the `--gpus` parameter to set the used GPU ID:
+
+```bash
+python3 -m paddle.distributed.launch --ips="xx.xx.xx.xx,xx.xx.xx.xx" --gpus '0,1,2,3' tools/train.py -c configs/table/SLANet.yml \
+ -o Global.pretrained_model=./pretrain_models/SLANet/best_accuracy
+```
+
+
+**Note:** (1) When using multi-machine and multi-gpu training, you need to replace the ips value in the above command with the address of your machine, and the machines need to be able to ping each other. (2) Training needs to be launched separately on multiple machines. The command to view the ip address of the machine is `ifconfig`. (3) For more details about the distributed training speedup ratio, please refer to [Distributed Training Tutorial](./distributed_training_en.md).
+
+## 2.6. Training with Knowledge Distillation
+
+coming soon!
+
+## 2.7. Training on other platform(Windows/macOS/Linux DCU)
+
+- Windows GPU/CPU
+The Windows platform is slightly different from the Linux platform:
+Windows platform only supports `single gpu` training and inference, specify GPU for training `set CUDA_VISIBLE_DEVICES=0`
+On the Windows platform, DataLoader only supports single-process mode, so you need to set `num_workers` to 0;
+
+- macOS
+GPU mode is not supported, you need to set `use_gpu` to False in the configuration file, and the rest of the training evaluation prediction commands are exactly the same as Linux GPU.
+
+- Linux DCU
+Running on a DCU device requires setting the environment variable `export HIP_VISIBLE_DEVICES=0,1,2,3`, and the rest of the training and evaluation prediction commands are exactly the same as the Linux GPU.
+
+
+## 2.8 Fine-tuning
+
+In the actual use process, it is recommended to load the officially provided pre-training model and fine-tune it in your own data set. For the fine-tuning method of the table recognition model, please refer to: [Model fine-tuning tutorial](./finetune.md).
+
+
+# 3. Evaluation and Test
+
+## 3.1. Evaluation
+
+The model parameters during training are saved in the `Global.save_model_dir` directory by default. When evaluating metrics, you need to set `Global.checkpoints` to point to the saved parameter file. Evaluation datasets can be modified via the `label_file_list` setting in Eval via `configs/table/SLANet.yml`.
+
+```
+# GPU evaluation, Global.checkpoints is the weight to be tested
+python3 -m paddle.distributed.launch --gpus '0' tools/eval.py -c configs/table/SLANet.yml -o Global.checkpoints={path/to/weights}/best_accuracy
+```
+
+## 3.2. Test table structure recognition effect
+
+Using the model trained by PaddleOCR, you can quickly get prediction through the following script.
+
+The default prediction picture is stored in `infer_img`, and the trained weight is specified via `-o Global.checkpoints`:
+
+
+According to the `save_model_dir` and `save_epoch_step` fields set in the configuration file, the following parameters will be saved:
+
+
+```
+output/SLANet/
+├── best_accuracy.pdopt
+├── best_accuracy.pdparams
+├── best_accuracy.states
+├── config.yml
+├── latest.pdopt
+├── latest.pdparams
+├── latest.states
+└── train.log
+```
+Among them, best_accuracy.* is the best model on the evaluation set; latest.* is the model of the last epoch.
+
+```
+# Predict table image
+python3 tools/infer_table.py -c configs/table/SLANet.yml -o Global.pretrained_model={path/to/weights}/best_accuracy Global.infer_img=ppstructure/docs/table/table.jpg
+```
+
+Input image:
+
+
+
+Get the prediction result of the input image:
+
+```
+['', '', '', '', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', '', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' | ', ' | ', ' | ', ' | ', ' | ', ' ', '', ' ', '', ''],[[320.0562438964844, 197.83375549316406, 350.0928955078125, 214.4309539794922], ... , [318.959228515625, 271.0166931152344, 353.7394104003906, 286.4538269042969]]
+```
+
+The cell coordinates are visualized as
+
+
+
+# 4. Model export and prediction
+
+inference model (model saved by `paddle.jit.save`)
+Generally, it is model training, a solidified model that saves the model structure and model parameters in a file, and is mostly used to predict deployment scenarios.
+The model saved during the training process is the checkpoints model, and only the parameters of the model are saved, which are mostly used to resume training.
+Compared with the checkpoints model, the inference model will additionally save the structural information of the model. It has superior performance in predicting deployment and accelerating reasoning, and is flexible and convenient, and is suitable for actual system integration.
+
+The way to convert the form recognition model to the inference model is the same as the text detection and recognition, as follows:
+
+```
+# -c Set the training algorithm yml configuration file
+# -o Set optional parameters
+# Global.pretrained_model parameter Set the training model address to be converted without adding the file suffix .pdmodel, .pdopt or .pdparams.
+# Global.save_inference_dir Set the address where the converted model will be saved.
+
+python3 tools/export_model.py -c configs/table/SLANet.yml -o Global.pretrained_model=./pretrain_models/SLANet/best_accuracy Global.save_inference_dir=./inference/SLANet/
+```
+
+After the conversion is successful, there are three files in the model save directory:
+
+
+```
+inference/SLANet/
+ ├── inference.pdiparams # The parameter file of inference model
+ ├── inference.pdiparams.info # The parameter information of inference model, which can be ignored
+ └── inference.pdmodel # The program file of model
+```
+
+## 5. FAQ
+
+Q1: After the training model is transferred to the inference model, the prediction effect is inconsistent?
+
+**A**: There are many such problems, and the problems are mostly caused by inconsistent preprocessing and postprocessing parameters when the trained model predicts and the preprocessing and postprocessing parameters when the inference model predicts. You can compare whether there are differences in preprocessing, postprocessing, and prediction in the configuration files used for training.
diff --git a/ppstructure/docs/imgs/slanet_result.jpg b/ppstructure/docs/imgs/slanet_result.jpg
new file mode 100644
index 0000000000000000000000000000000000000000..011857fbc2295b91a96d938f861d38b8e07421bc
GIT binary patch
literal 79603
zcmbSyc{o&m*!PqqTatar6rrpYvSxlIgrvwG^F#I!!pLOGzE21tCVR4Fmu<2qBzp|T
zj7nK&&~%uY<2^mkbv^GN&+}f_`@Z9vxz3n5=X~$)y?yS_eb2$ygC*DrV*?`t*r7u(
z*dgc-c0hpX!44n#`}q5M`0wKg%imYlBS%<{9A!Ov^go_#?8lC>v9ldLdW`cJJICJ#
zbi~EU!S(me-;4a`sfSrvSXepOjmURRMjr4YiMfeUDr1-ykTT)
zao5ty+Q!z-<-V(%yN9P&;G>}65LD>n$f)R;*tqzFwDcDlnOQHha|(-!OG?Y$l)tU3
zZ)j|4Zh7C@-P7CGKk#wz)A+>X)bz~P**Ow<`P=s&D?fj&Qnz<@f7AW|d;5RoIs{|+
zPqCoa|4}YpNUp=LmqhqeF;cZk0Rcs>X#M-sw$tF&paO(~VqN5%6_8khI(ukU5x(mnUS&X`-4nSmraq-IeE{>q0SbQ@;_7rh
zPtjZN={Lb191B|8r~0C(OP@kcq;Vm6`s+ulQDjEinJXKw+^i3=-0SCbjSdEtAmI}a
zU|qupFdiM#B9!a^@!i(?RKc3AByNxbrdOhNOq*IKzkR-J-zS@(_j&fT(;-dCcEU1o
zeTW{`%;4Q}oWOhGBg0jxWqnL3d&+1J=I~FDo#IpxLEZPr-~DGt
zweCbSukEacn*wVXVuOzoX#&j9T^Nr6ANsxUVR(a3vJNjbpU|Dt++8tx7;vL$bFRrN
zY7u-hLbb0Y7tGtQ-Z6WqJ0;z6^!q}BV2}F&j28)sQ0Srn$zX#8z2KA3{}5$=YQXDM
z#0gAS$N_ArO;;34&hC-j*l!BC
zXJur16K+4oJi{y%Vkm(jeQcA?Kp>qR{4BbjqdIiF{&@PbgD%-r%e*ZrpEf$U4cQs#sFm7WvR~}-
zt|RNWCD_bfI4dSPw$Ljqk?Juo1ji7Z0yeFe647`6hC}FBIzxr*Ln#5^W_kNVkLC|%ZXf7Ie_6N
z=rac}Ne_l5r0HZ#*DzjKsw0b}yncnQ3Y4f0_G+c0LI!etC=Jh7*w4rIDSmHei|JaL
zks_G^3hoT?^UI^j)bP*YhndguZVO@kmr&t0w{~Q5~wtdamv|C+@Y+oB%RjM~|Dl
zyqaKWdz=w{`l3WOJqXRkEJQrmaNHvMG5BhhY7#*SM|<#NC$tsuMbqtyA*AQ!;{YG9747`VNn_13ch%7EbCKuxk^kyNs3md--q$rriCJC-*tI6YQ
z4`9VM!mW}7+1Zlgnl5iHto%E$?rT^Tnt$p3d*pGTAlG#^NcvW^*2pyuT0&j0wa?s^
z{)B5mDbhUX;oTMW;4gRd1@^MPqW)}El*vWTaz{;~_{^;{#Ya8vwcVzjQOEQgaRF_2GJk0VtVxwLQJtD$`;$
z@d5N%?41_W>(JWm%9Molx3?}`t7WOJ;uxjE;}p<}uN_fY77;WX1)s&x<%vgwTS_8#
z@5e1Q8)mxgUbk{a!E#`XOLYgZ$Rx)|IlIlT~#&E3SW%S+=7B{;7O5d|9PqM5?W%iGvvShz?S?(g{>_JZ10~i|%_`y;1OSmdHkEEE5M&`&q
z^Fv#D7tc3kRrtN{NFBGGn-)=Qtf=>s>hjZ_&1aV~xNVD!@=Jy8oPvbuDglACN0dyNe@4i|Tk%V4
z2Qa~3ojQM!Ys5hw%;X
z0d-JnU_TbDUJ5VUU}6!{7@xtRupN3>ASNuGI=MGW!>;42!aM`*n2LXFm>z`*@<7A>
z1t{IH#4^jmRWLmo;0Wdf-opQ0eInJNp`%WMoWPx@&OWW&u)Ey~IUJB6?;
zo%Iauv~)^$$are^ypBl$3ey`wu$b;#@Vs17_R?oF)rzb!YF&`Vw0`_PVoHG%d;VQ?wU0<64QfNe-v|YIV+I`;lI`jM}l{Qu2HIal(R9
zIj$T|(c1#f!+UXnoY1;IhDsnGz@Dlicz{GbYTuvOpd*y|^hEhlxy9*K&E*8Qv>S*J
zRO%(wI@bbsuEMLLPg0Q?NQ#HYD1*sU{8ibVI90e9f0S`g}1mBye~b%|7tnSY`t!K1YC(^?DL{X_tLYhoP8|F
zeg(1tH|F~Tm?xevK{i}xp@*Vb0Sw7c00r+lfD!zZpEe3+JSa&ae?IRcD<2xhwLbLb
z8lfngi~C4wYyCxDG!n{8i2%}t8rzGnwWT(9DaYqDN2+3@;U`q{T>7+lqJn_k+*8JX
z`m6)&^p(n8PWC>_Jg?5D)T?n)pvD>U6?<^x0LG2Dj|Vi1XlZ~LB|-it8sS9Ad1CPC
zl*yg6O%L4d7RryJOClvQMgp#F&(8rx-;QaJ6R2-1Qny
zwwdnWvXr|JkY07Jt>5(17o*Oxc4V>{VWKRyV&neayl-ArrTu#NAQTZ9g<%Ey{I}Ui
z)*mo{Mmor;PWMJTB4ZC=wKMBZln~1&!CTh3J)$yYilcKb4YhRDbVIetxyEAiL4NC|
zXT{s3ll5t-Nta}UsOlgGBnI1Q#;GaBnL_kcW@P7)*?H8?fcFxSor$|GWb~w(_aa>o
z(7FCgOnLT)RVYWH&V!^+Jp$A%hS(4#7@mM>!1oaF5@-S$PbZSc2AZd)wAHo>Tnciv0Gd|ZSUxU->%oY0l=_d8BO
zNVtg{XNlR~yVG;`TVlBx@A-x_pt$y;>E5iwWgZ@TzUXLMtGv0+WczJhAD{;^pFfZa
z*U^WlhP^y&3?%b;=Mj%AA_ws*GEOp3yQkm)_TZhVT4~Oe(8AgCPFYs+f=jL!ldRvO
z3_(sk==3z;4Bef<1+YTLhnb0>01(oH6xFeCE&gTQ$={J?<@YH>H>9_$xK3HX9SQ(R8w$FYD``2+Noc(a
zn-Om*k7e*HiwqL%`;2>i0qPoYF*H@t1$1z`qpG32D0`Cy+iOMOC
z@D{+dm~nedP5WGfgMov!u~uJ97PEps#N2kE`Ie^Z#mclBuj}I-YL|}7k~Xi9EmN0K
zA%~3f>@+f0jPp>FiDfZWR!WyLC5?1F6hj@e8uM~cZv`GUJgM|HZCGxznl=hX6epZ3
z#OLz-d#*46Cf^n0myBs1>8uDmnn>pWoL2{xPWdQR>#7=ZrC%403QqE@
z3>!);L(p!8RsWXg{5EOsw>3
z1v|KR-!%+d7R4ube{l7QW~85So#$D^@#8%q5Qkl_r3m#4!NnT6C4f1GEv2M7mpml>
zu9JE^dMVlZcFH*2yyIPcv`=wsvulgZsWaQfAKl{Hvw88=wc%0^N1weiMV=oF;7}8P
z+f$Gk3el;l0JH5+gNDhk!o2Ia?1DZ@1O(&$k(JjTLo9o^I3h^>8tU6R-nW
z;XE^6=llVTePsZ$VPa$Rvj&xi?=TO5fnG=12}PZ0dhOnfQhWcs1$R*~F?qc18sSx@s$dLO{}Tn=D&r#TN`
z5Qf~)XF?IltQaGA06X3(jp^!w3o(S8sInwwEEFLy{HTz-22WIwl+6lq+nRi%^?et9
zO1NXBw>uZ!6@YpOho$NX_noy0tO1=_2y>dI@VxT|2)z4HyPvK3U
zn+s)iLSYH(g?y7N5eg49(YgKommXa+wuPpYmxk?$F|8?D_c+_6+r
z?mi>jy~Ce9X!D0s$qrya?q%wI96PkwQcUC42bv{5@h*@uU35F)l8SU@K~?mF0D)6>
zdV^~X?l}yC%JMBCDl(=e3(+pxCMn5|UR<9VueQGm6cS)%J1u5pmJg|0%qiF?iRokvL)h1DK&T6-uGz
ziL3k5crJv!?r~__o}-=4+C~+Z=uV_
zf=~fai(5Z{kr8t)%LD6INo_nFfZg2rAd&UXi(=*3-}ALm&Nr6F;}y?t_X*2#aK;2*
z=jtH^Lg45*LkqFv^q7d;KX+pH0QNzS&IcaXPreUb_X5M&jeA=?Hv3^O*8!|suQcUh
z=;{-^xOJDZdc5u#Ocy7p14-RxC^YVkDSNSNh2d=*|V$A!_H>
zvQ}2YdV`7NrPV>4uq%9Q_`lIkKTGs2M6nB~wT|FU);9dU6Mr|wJ@wz(Jk7eHEMB0_h
z|D>EwuvL+nl9XieJ68pjI?8b1TnJNGISNk!$_`+1Yt#^s-6^EX0n>F4<1FvJmX&*}
z7BE}3hjW^qElKqF;wDn;}{h+PeENw*H;Mplr5)ZSsYVr;4fhun|1X7_9GZ?0Y$m|d?jFwf*|v(
zQ(dTS*C%x>D1|emnM4_8td6apyPscZc$(+Lyq*jBMxjgHMnj?#4RfyeLG|Llp)HY(
zkYjo<+?k~ZFc-gBBF&lkTK5dF)eYx;gd}j1AKXS?pFBgU>T8aBJ@dOosd_`{JD&ca
z9F_UY>%knNzPDD7E9~;RGcmNgQ=O7^#Js{p^3z2d)THO2-lbq42U+(hFIsRbvHIhW
zn#`HrhSkZ$=06C!%XcCmJHrsyrk}!esfELEvnw8QCo1@*nlq
z==C9SH(Svwx*AyzE_9u&ywl(2WUi;JjbA{#<6($Pks(q*xwpfG#Iyd3eu7XFw`+QA
zjn5N21K?;#tr52rcIg!T2?t#xnP;}P8Q51G6|4cBeRMCg3MguKIU<~i$Jfj+QivH;
zzS~XB?z%TyziBD9+N+%3yIJn7``Tso9~Iu@_i#9->p}yT!vjNkB)ehjvh?+J!bm$`
z<)|FtU*)-ysYmC1&BT4xdCvRM;zu8^lqW9n@L$N@)YbHPEIg;-N$zN%rr=eJtdhiC
zg|6k)TdH@)+e#OiWeG%NoZBfAob(y;#g5Pr*YW=y>Kp!dXCwZsS<#z8Xdl4*h=8Km
zh9Oj*(`Ixf0By20gU?P8;|So&9c%zS14?sWdgzsJ6ty~4xm3@a8}fspzN!6hO)^gh
zzz|{#XbU3i8_eRT^UpEU!j+&6QSjZArE-OI@=^^Ws&y}n_`EURiLHA-k#lbMMN2|}
z1L<6wtM!soEc_bu<#7D>vrs*8tA?(03G-{YM{PG`Lz{vj8MM)j!$%RdvOQ+8T1vU4
zdSs|b(N&7*xut)9e3wN$v)h10x^#L=^*yJij9LBXTgxpoTe}qxm02GuUCky3+#pkJ
zT4u`S(t4deHCkRY?RR+!-g*X&FXJ=t~cRumTKIY6jW%=bTN3Ebd2akKnuh6smpw;DrFZ`|V_5
zPZqEtd;!DO$?x4M4P2q~p*TsOlbwlzx=J`+pm+8cRUyq|?&gZ)l&>^e2-E0Tl3ZN%
zs+pEODDoCZu@_4Y0tFdID66jRDperVh&f(~4&~$fZc=i}MkD3Um);rw*@~mfY*+rN
zsr=lTKBbW;Uy+1%5*!=l`V0p!JP%P6H|N0-Zhsv
zc?dZs5bj47WQ(d5=Sy>+z3d>)U#rsBT`Xl)WrKJA8K6D=Y7Wz$u|7_yor%f_iZ3@V
z-g|%7r_i3*T<|{x^&(IR%iz}rlZ3jF02+#Q89`gS*2}Sl7)^vYV(BHEat8rOhjjad
zko9p4epx#}fKs~Lr@yY6r%lcP((H57P6{X4xu*qjj!Fvmrs+Q@DQXJkFcdd*e+r*l
zp`{^qa5Zr_Y&;r5TKyHjBf;a_!jLIFK%M5{qd+d4$jdU4ftbu~B5Gchw;bnAuU*RfMb6G^(b#D=L^x-}F)6|LG=F8mGeUgTq~@
zdgP%G?=dI*4P-hV19LCqYFzXbjpQT-9etZPD!-o(GfT`ZT|j;wf|VrX$?+QLE6(o|
z8?sz>j)60X2HYJ8zwH6>qc#7z(zQ+aQ#i$t2asz{?9^pQFl%&m!EtMx1X_DZF?h5u
zpSu|keEf4^fmG1eoHkBd*@(;BG2$*x88bltEaDJhH##}P5K0Hx2aCdo!uf#j^lJAGnQ)^Cg28~?!{Kf89xluoz&V3R}wia>^9|>yL<=_{;QMwc?(dqX7H{$8*#%;idQcs(qAbO$xAgayS2<<>p2OQqq`_7_DtNrqRXz1citX_;fs$`LhQK%|)iQ+OwquXrfJPS0vy*C2Rb%UEwuC=U
zSlvN6lR^0(^;wS7lKC2+U+Iq#`Pv`UqKB)8BG}n>M>zn@nbDnLi_hnctGF%;@x?ui
zV&>)+7>yX>GH;!4Y`rz7t@a3@PI--&z}?p83v0aRHVJTfun$Qw#{n{!9-MIa^x{co
z0|Y8=0MV2oS_Cz(wB+^Aeqm)b)`sRuHIp}sAlb*@ygB7b-E8lesnlNW$3?e)+;(wu
z5Ep6-ZQ<}#_z@I>_L!N8aK$HDpxhAgTHzg3Q63RgD|x_ns_MbUce=Ly+L%{J*4Vs$
zK$GwL<)Qdx!??@dYdF9WArQ_B*wZ3`GO9KycA$nA97<)F02sEekLVzB2l6_VwPQDH
z`rV=#|6FnE>aqzqmNs%dkZrLfLOJV)&pv^c260)LLmhGl>LQ6_U
zyPufz@V~5Tx6pFXJN3JCtF~@-$ld2`R()ua^*Yt}+B!jK9hKF)#o1D?k?H>O;gq%2
zoid>ZT_s+3Nyeg2toP1{{Rweb6BoQ3t$V}DYGyqo1a#R}?m|$k@xWPpFFq6-NzPIrf
z;g(ZRrM9-_v_Vcz687l<>;&4B1m|I}P3!o1Qnq>@Tg6gOwJQgy^%uE^gmPDIb88JM
z+N>J+aDQ)4Pe04`cBJuMQS@F9t)qkLhOL0Iu$hHEP_dk7K*4t7IZ)Vc6#{C_!dCT)
z9-ZfItMYyFa!{q4E0X6w$d?irXQqC0Tabp?$%Jw~{npOo%BKX)PQH6?eD=j9
zqmGNJMII+L?o;p#K|PRrn{vNXJgWy&hupY$h?0d>MthRBRQlD^5=|ekYqWOu=F2!q
zeL9=<@X1H?uW;m|?$%fMSNJU-I87WdHYj?#8ln#8Zm=I3T0mOm(6+aKcniB`+RDBU
z4n4Jy9q+cf5RGo5tbrUZBn(u>u}VU8g5Rvz!2T-;$5aR5VuT(aEu$V@+}@r*=2~RS
z4auSbvO^VkW0%dFX}s7}$tXKY(2y@ndMv*Nrm7
zA*g=j4l_|l_6@*E^J2clc)*YU)XJ~8T%r`O3Wb5-8I>E`#*kTikC)`MecT-TAa!jd
z?fsATHQFNxR>fnS5HWKr{lueaD?pZNVn?y=Lnh>uB`a(;RM^WqT>I+K=X#Aay>}q;
zQtI5G>O_QRS&^xxSN5Jg)Zf<_0FTu#4V7;;n9T+*7m4NRnurv2PBhS8&MnZku=nSN
zl#Y+(YjHhNBjG4|2;e+oJ&TeEa>!C!`)5A1E;O!euu)}>>2h>d`3e%vc4`{cn`$RZ
zqpM|_RIV&UiIEe&Tit9(dQ)o5)Mw%t4lYoM*>xSN*SFy4Th!|Jmvfg+xpez`Sj~i=
zZciIBYRc-KuD|upHl9oV$BoU6rYpl6`St>o5QgBUuYJ+KR?cxH3-)IQIlk(v(K;pEJNZ^@Dy>;vUm+|w>gCizR~78a
z2Y=fI_FH)I@UT`H^V!+A5bNsh15CY3z
zbtevBCRJ<1pDZLIpveQ?ZlmUc--*l{yPu$b(mBY2WAIDqxO#O4-=8i-txzJujVB>w
zEptfmLtm)6d0gd0kwvbSqISs{`ROCD2Kvxa0Vgi`C`*6O}%XLF^ao&KBn*A4k%CRK`(9ut2Q1Jyqry$uOKi9tMlC~&_RQ+BE^zgESr)CI}b5%cOCKDk(_@HAkf?q=fX>MVvh}P>Si&gvKYJ
zjD0H?x{x|>0Ha^=z|?M(q(l~?RC+=n+=Zc5dwQ7mf;FiEFHkGotliRfJGr7bdy5t*(;)y;JUYXx?T^uz75}t
z9<*moZZs+X9%dui^is?9utDw*H(kVj-!6o83nZZE$Ke7RoV=d79k%c--ehQrRn+SiuewGvsQU3$XyXEt+1AW>-{bN}0xyC0Lq5f^fijTMjUx`XbCS2B7Wt+sb(|=6
z+grU>S}QR=H9l(gA1l!&;zH`rjiZ(F)fyTsJuh|nIKiHYI-~jX^|9S6drJNsSRQ*e
zt?@@C!jmR!=g6VXE9?VA8oJq&(9~>p$4B~R3BpmE{)C|UtMTdt0b<10%NjYbli2ZuUCaS-RG(+AIJR72!KGhW~eZ;b~-+QPJDs^-FN(;U5?P
z0rioF`v_{F$CL2`m`Vo*kV7)Wo}zhtmd+tOeQt?)^GGig8Ebb_<5qTP&GeLQ
zZ)~Tl<9u!AZ^fwk^QO4R6(rA^MPjL0AGFhIz1#IJob{H~O4aGQqUs|{P?pVd%lbMzrCVeYeGf>k!$uI2eVpjdVmXZ{U
z61i%{^8@!MW^PseK@_}&TFx(dJQ&ZR+vn)`LCy}RyFCL(zz@Zc<
z@RdGsgFy3~A9#!#iepO@wtmQKJC9J1-5U`5KxXU3Cz-rio2jXq*DD(+{2JQYS<`$o
ze)N8lmtlrl4{)`o5mlpH6_=KIDMQ`hZJ9@RgV*4o>L#Ja
z)i-rV$B@fqSYD3Lgyg%*RNUv^goOBs10F>E|K#d5Mnm@ZLFcL
zX+NlMSY4$z7300K`Z@OQXZNxa5lbX2ASs09=`C1B&LWnZ-r$5}C(5Y9
z>d)tNLM&|b_^kp6HMAcJXXEtviZeSc;-zc@f1Gn7K(6`^X6YG{0s%^%dBAa?8^h%{
z(2Gw-h&_0eEb1VCuVq3(wM|5+crxgWPwA{`aBYugkuCA8%h2~UDJhkUE*qG?)rze8
zdh${YAUl9Xhf!E)vVc(0TIVUz;0jF)>n~B&oJ^e08~dTK(ys6((uQnnlWkRD>M_iv
z`R*L*d(=0XedvlqAKB#`x9DZkaN3A>{fL
zCjuuU5_KgV|y^IAj@sVLp+Z%(~p2G=Y#xKw4+Ez^Yn5Ssi~eTb3ZHAZ-b%!)^5Sxajr@KRlkP>
zQu@?Dmg+6zcC0AIK)_Lr4EV-e?a6k$8j~k3-!s`aFlUw@(mVC=g?*jl)>OKw-{+}}
z#!FwPgS>B5*2us4S=lj#nS>@Q!0;;;J6P~IhM5C#A<(d<88>GSu^!)jC9uD>EEbgW
zZ0#>A0(|?bxAe2V265^s1XF+5Qq|@UWl9HIx;=lI+E%mD30XQT2(VDbw+4{hI$@BN
zzM{s>sMT=XLnEjm3>kbwN@HM^jpsSn1fMhR?eDxRU@6;vHglRfeO%vKkHW{K!`U!h
z9Yo4a9zdpM14Cy8+MwRsMh7LK$8xd(lQ5H`Dhie?`r-47x-mdbco-@NTwvJ|AF8b>0jFNYWW0IZO@n)ad3TgKd
z(sKE-R5EtTedv6ZpLZBuc7E?~|3Uzgf&hfjgTe>X!iQQz*4@deyGvW|^KAD6|DR@E;a
z>@6M^{)F1Ay!oQ!Q|0t@+qQ$5KQuco)$Ul>wdPC6u(DY&Nqu#BOkP95d8(9d;@-9
zYc>&NUnl?~X(*tr8j!zBW$SZLH=I(#9CvIiS+ji-d%NwTp}&N5q>9}?wuSfIEbpHk
zkg5m|0L;K+<@BqVu4`4G3fdWfSS{8w(`s#7PY+_ASehDZp{FXXX0C0^2S0aZxAZLFKL!=W*MRo%4Rq5Hnmb6SHD=ox|r^yRj-GUuFM)jd|j~2_RZj8
zz1ze%?9H~U{+FPK>IAwb+JITSiv!M2w~;&KNw^pvwY-2)$lFU_Z(mxp>kD!{5i%*L
z=2_U%oUYL2oZ@u0cU|%R#&oJ{rI5;Mu
z0n}6QAm<;7d{4L-@LoElglf|LIJl%xMYoyz#bs;JnB*X+4OY@Pd_n)ct9`wg0Y|O2
z@xP1*l+kr8)y0lkzWHbrP?BArDe;70=S)-sj(gNQ@G>qG5&Pi5gOxh-dN{A_+z*tJ
z^!)l8osmi*SOJ3!WI>*D0g$Ey(y{I^a)+ON
z9c45IPqH_9d&*HvEb$o^ZCpP*`AkbgjC^!lf3*bZ-@`_iw-A$NMQlugMm#00`2cpF
zr^+3cirS{;yn17)>vjB6N%9mU(A(Rh(`MR&$@s&P&6Lbuh}
z)>*1$xBY;|vy*}1VtJza=FdLXs7{27f}etQ`6hJKpz?!thMDb}ZpN%E1%Av(a$mXL
z9I!9~AtFdM$G^7E5>I}6cl+o>gxSY^Glp0oT^1Ef5{mBB1l&PRHrEPNVE=wpD<<;C
z7|4EaVZmFyHWUmhX?&%Xes9DdK3k3cXeD(v{CaLi@MR4M?LhGu$$19opg@?63$aur
zD4|5b4cMSf)Nm=P#MmSHrG~|xRgb^@qbBG%o1mi*UFeY~d}5e`g?WAd0%pt}G8T4u
z%=YvBX^3#v6#|dohv`_dSvUOa5ZNcIms<#$B#t0}L-oJ%bs@8?NesyeVpiw+i(Yx!
zaCiK^1^GgKm!u7$;yuTWON=5E`zD5BModQ3nyo96xV@K>a1>kRr;Eyf=h6Bgf2f%t
zq^mrkxP^w7ji}tnxq(l&ezj=#LAdw~*-*cG>)e~fy+X{=Ja*O$6p%@$_9Fm!>D^9o
zDlDV@iYmN5zg&8{hTlJQ+|1Lhx&O!Nk+{8NojVQ2@0~v!_WlAwnW3u~b6i)0p*aRB
zF`x=OqWl*$S2h~=zA~
zZ;Zg~W@Hz{7v-O9$#|a0;Fm;X@bpf>SJNzGn3k>--=*$b#xIX#kqtf|#Kebw!LpaX^F+oLwb?SDf!d%O-M)doT`c)*N#TPS%7fAZx)sDYpZ)BYEMeS>@9vzfb9ND
z*i(RBH(sQ1R1C#qM`1WBFpG5T$#5RR+T8ujzy;4Ni{9x!x;J0nNR*8DC&=p5F*eE0
zLRbp>$}}DaA!j-ba`A}g;1IFSx(`0Lbq#vA$nc}OMN?=F5{e4?HQs`$-W#D$ygii$
zF^%80qqqYuUArM(aQp7Jnhoo9MXFRkP6XkMh{OnWO06O093`p9Zaj~-C|E#4k?58i
zpL6|Gx7Xt>`5vVVc*(u_3}b|EqKuz3h;v^0hS<9X&DhhkFu$cHUdm{1RY-S5wU^dt6576G-l%wWW~{h-akHz26mr>L_(6M9
z+GT=AHX*ra1521#ij7BjB4^e6;V`t~5{3;GvDG*I&a=vOnI2j)UJ$=2xuaBK5|ZME
zI-yDpRTl3=-{DV_=`Ck`hOG4*ehG^|iCL<+4AJ)AC~Uni+m+DlF82p@iwad~Gk-z?
zw0hWbWXu$=>FK0hs!U3#C97Y0iYR5M3~8_=>PYqfizpIB(00fj)24wsF2rT5-^&t|H&}P~;)+y|iYOkLOwK-WvC`u;(_q$#<3i
z&Zl0Z)AIfj*VKpZ%tHHCM_q$AH^(2qoVbN(M7_Tq-gZ(Sj_%CjlJssoQeu`l$Y^503AGT%FNhQ#S{3tiKl|Zcp~Te@NwX2K-XKj
z)(rgxRkuW_yli21>P^CVlK#_=n^L@)?g|e02lCW(kXw?OhHzI5qlHo78zt3fh1w1v
zuMll8rK4Gddb{aalGby}NBL$(+A-VVi#KlWl;_>P>5_Bf&w8*PT^$KuCV#41RPLc{
z9{POe6NS>PIij!mNkDEgz*w=`-qOf%)UZ(N65dlsyGlE3EB988^ge6&RcI#+bUZ0#
zB(Vew5Yy0zdK~-WJJj62d
z#688+_Z~LXeN5o0+sjzu=4un@d$}r-jC763XAu)^JQ?ulm)1#(SI#e9y|Q|H-eI?G
zEW6^BhklpCj=@k{=fD^Eprgw?RK1jjLio42AdfrQSqqoumMoj3F^
z9W1?;lRQeRjF;2*K7aBo?*}F66iu;zEyNxxM}Y|NBQy;eOPL|#HgciNk78ysyek;8
zlQqAUIh7;T7F4c9EZ=$Gkkjk9aCU!$6l`K2pW&R;A7Nyt*Y>~(A&PfV1|&nG5I%H#
z3w(b)sFg(K}Zl2Bw
zum1>IP5Iy*z``fOh@Vx^Wuh|hdq$(bE7w4@GJ_qHz3;z_O@v0+d*Qqch)N^Z015-^
ze=t-Nk~Q}C2C_?6I=xp*^)lk?<5z+9CIt?5dgbNWx}PXDb5{Bot#OfrrfEZ0O^wZ^IXv~yNC@TfU>G+hcsv~AE|GAJ_20!7s
zr5N3tu=#e!V^8SfvHR!Gz7y}?4#nH{h$}BuTRi<-E3SyU+YeQ-^|j^Zc^nPtkp)Vj
zfyOT^$&JSQ9XiAE*28a&b83tXbg0w6e*Z?4%t7o_GEV_Q9n^jQkXhBPPY;)#*FyAIQ_7-sbaVxQB#c|f)c(S%b&Sh>y~YoX6V4pG
z#+(Lwxx?K!O0Z!sRum-$vImv{F5PpZvk%a&RBNL#h7sFr5vIvjx{CI?);CB-$H8W9
z@I6QKW>b1&p8XC)JsrRpqH9$9rP6fDDzOjW7@Vja*D1v?m_W;y0Ko&@<2NYGg5n%9
zsaXBAX4AyR=H{ZBu*^b{?S<yG`#XxD_itHH<6bVbj8^Is#5X9lf+CDZ>#3(vmaVtZSK|HQ{RAsGpi4j3-5tb
zx&qXJXAB7BusX*KnY2=4W6&m)d``1xT2x)$jQ`E**)E6sx?EThn%Oyc^Lv$bk8^Ktw|S&%w|$*W6JcG+MO~kHBHXe
zH`*>RKj(scHR&1TZ}p!LDXIUr4Z#XgM=x1_r<{Qr5f|xDXT&*nok}-EpJ!I$msT*5
zNC{m&bRd-Ij|J5n^YX6r0{MX_Z-9VB&X4FTl^_1~2oeh2JbPk*D?e1*U#=H1`2+vh
zn-T_ccLNRu;X_136|dU{;9fM2!`vTDiE
zbV_mIPrDY)=;2bpx2x@qj&tN2Mv@h3ln34AUTO(*>3B@$wy#>eS+!33){&=cHDw>L
zU3NdW-dDa}F38m&u)O$H-i%viRonXCUkY+Peun&vWSrCkImYOGkb`L+!0s!b0=tQn
z+h((V1CFv&N}xQ^D`z0am12+^=J~*U#;l2cWvZzph*NslTJVx@a^c=F$B@H9yGZbO
zII{#qKIeEpLb@K8#f3w}mTp-k%6G3y+3w@lRdu&>)l$9h+`Mmas@_cc
zR3##P_L1L9gfkqF{=?uOp!HDPQ*`cNplv4u1X4DGkLul<&z3WO%FGsTQZ}xf|7D>v
z);8+RivlMRi)M}a1@C=q_dE~Fw@z_qIcmNPZ?{{cb9pxqH=287fv6^aQ-Q1$tmqED>*U%+-JO7KVHxGyUjpBw?3Q@9WowAlS*$SCfk`R@hsgNxs*~d(g
zeVGu7ACr(|ne4@6FxDg?WSy;2mdsejH?#EKJf>mFx{ywz!`u&`TvU5nO>83Pu!C;|j3oI-SLO^i5_~wI7VnRM)%8&$zsFW_HT0
zr`**-Q|!VaVY%eR5B(j->mQ$1B*?C7pGlb^u(+qIA(jJd58ymgG{YZt^)
zYiZKEw#;G2_YDtU-rbRRW;g!QXuDU|6MXvbkKGJg;SK^rfDZafy)O>#wvypg5In9j
z$6xsc2PT#HzPb|8VptRS>V}8L((01Ne8^aUGhB7MVd-CYSibO`f8AmKK5Rh2kH-(Q
z1AICU%A-#F!vE!&g*jId1PM1VGp
zf{dbV=h%YL1(w+w<+-0(@4yl7=}NfJSH<=?@CLy3&}qm8y*Z$I>KP;a5XNtI$QK1;
zNG8z=bwvDzspp?Md;D?Oa;QI}{6;hT46Cg#*Zo$$*4~oqqS9iF1`3)j$;aI<35~HV
zh%62EQBGyeCPvZ?s-+Q|N7Uj?b%S~HM=nI^B92qLPQCW5e|N)4`0b9p7bQ5dxp$kT
zx^?*PiocmEmffdW|57zFAvxFSUZ0q$sh{ON`^x*@6m(Ixah^9~Af-=O3OXekLnKn#R)qo-PK}1Sl0@bP6sfzyCx7sc
zl-$&Ko6G*=VE4VN{9=CJzh*05GN%c(9~5Euf7wXW
zt$qI;A8K7Te_9(4d8aNRj}5Eye>^NvY>;9U0&f5S6J#RHjDrAxA-Ibxi=(u}2PsZ$
zYN-R#DN2_<@uVQ3XejsFM0OHOH9lllslLhxEq9!D|3*~DZvrI;OKmZEW@){%0Nj4!vg4hV*ZOhFRv2Cs&-X0+0vZ@(Fe<
zJ*FG6Pyg3%Q-l3=l))kdk$X6k$~{?+I!`Nc3_!Yafb4
z+PMM$8c!sNqeLU)dvuWsjNxK!qP{|L(dp#z;NR7*h-B12>!;pSCLz1=s^kP)~|lOkwrIV
z4}?DNJQxz3f^RrS_pYi(l|f%{nEXS58U^QBvU}d3@}%>22IpWCI9isPI2e{?EP;kQ&eqaFoR%1iiGg
zSRTVXG?9?cHqgm5wWZF;8+Ff1BSf4c(bw~@ysay|{62u0+85)YDP=1(;$G(Ae}R~f
z6U*lc*n+Rv1vSxynfEw(cxx6M8tkZNA7cdf9Bmp_@Z;)#Hjs4j
zvg{Q_&HOR%6fwk8<_uQ60WY$9CVez|uIT)svK>nqZ1Zx^50v5(&IZ<@4y&)kFRbYT^nZ0CIj;Sm2J>pv+$7%XiEbM
z>kQz0=yt>D=mYILKoMM)KrYDNu5F^gcP|wSLY4H=OHlrc8LMEeje`5XYJImv7S9SC
z^3#-15crXL&2lI8bg_4AA4`s{SO|5)!D%l
zE$cgGl;1sRtbJha9Wj+lgnbB32V|1YQS7e2Jia;!h=$=*z)S}!7ZFATsZIBDm+3zU
z6<}R~W~0A9bp`xVY$cT_|8$X!GSt7@U;co%)<1aKOc*i_VERv*$w&GifQ3Ck>P~LL
zQsz%0_Cf&JZI>IhQlZqfwI9Dc=_0=Oaq^v)gL`hzUfX>VuppA{^3fxojxMEK)L
zlM4W)^V!njUSuh8`M;}{{sSw;%tS9xtW}y~TIi>`4@|kZnb#;|{okG~DasX(|L=R^
zu^5Is8}#wFpLFN;&nrM)#X|VFrOWM2m~La9_7>+bEAhusb;_06Uxb8DhAC?trMy8K
z2AP54Zs!2Hej~bkD-rEK9LJRz1u_As!xN@P^sNg`&0=PA?Y7p~5
z2Oa+M0N4>SZs}-={`!{lmJ35T{S>JpEi%o-M|&!-?)`X4zg1K(!J)H@-jO*@rkScy
zK8%+e;~n!l8R_=S{lM+dfUWZUS`FJPDkf4Tf505sX~1LSS8rRO{yQ
zLh+RMp51waeiV+j1Y%iU0dc@u%+$qADtt@9x-7Yn$2>>3EqN}5a|gWK18fh55rFz2
z<=~FLJkpW1y}k2`gI(FaBC17V)wgFX-@hHL`I;22Y99AV_3^h|O1jstoW;#9%qqoE
z5Ipp+Zjv{Xde2JdT3&LBva!Q76D_iJii=dSzi%(U7{zF}goC~YL2u~t>nF#)dh7V~
z&qh@@KAD|+ci+w?Jk`?K@AE30T!~n_0a(81Wf+K;%O^HhH$fV}mN+K%k2m%2c%5$i
zhMwGfAG-q)hxL;9vky3;ngExD2Vi5K5X8BMBZVo}3u){#fs6=y!*G9VbYjZv{iDechyK6LJD4p(3r()k-jM@48)s
zlt7@?kJ13INkK7roN~yl=_@(o;0&X(ly|a>b6kN76dg!=vMYH1B4G!?n~**)2Pu);
zhIdr-3Kn@%_Z4S_R=E9g^xKCw;@+D=%$`{|=kKnR<>{UmDk|}CA4iz7weRysq}(SD
zonu6bQd&2qax6E>=IHP~nOt?X0h>}NfS7L)F%;vB80eX}`|;>n=U|>zo9l|R*7Wl5
zBq5Ge&jirdj1<5Cu8`tEcg0*EF?kdI5O}Z_xRCXytedYQhY_y0=5q#JUxy9;6TKi8
z;F~XM{XXMMTUVb@(85X`%OtXmSw<&3YFC2vn0gcXXiP!HgM8~2Mii7`nBhnEUA_A8
zZQXcqW8qJq39XuiSpXL$Xdf;(Z-3tM@DrpGOLd7r4=1hM>buqeZIwb&r?KW3`w{k4
z8VR@vSuKa7d;j?Docfiom*f=)I4<99Rqu5b$38mutDN$N{)
z#3t284qhK&D^ZH%nqFcTOeKE~N4;<@DgNQ~(y7$ma?R~sMtB2~@PFMyGfTkLVf9#`
zTB95z%$7Nbrn!0;!ZzK)bU1oH$G58xfJH_3LH7XWDrCSIB=2kblp8}?6@uCKOCH9ctT4Yb5wLZ=MZlz?Cjl|Kj
zWw1}jFCVW!wqvtfW~cYEv7E+3xa`Pq?H~vv16gKlm8N{+{87d!eB$oWvidly
z=3I}7O_jGgRE&<`B{a^a!yD>Sud5rD2nXyoKb~{RHrJ*)
z*ta=%+Gavc536DvZrvXFvt@*|W-&sji&*9ZDV#c|%xn)5mCuY|NX2ngA|uXERHOZa
zVYbvEGb)jadFFF6^~&M4s9>K$`)rN3-o;1u;V<6G3C?TMa@1jTwyGKD4KZa=B`<@IJSwxA9%ujDm+QhHeCuYu^FQ7^*eAQ=E
zXX$)M$JD!-`$lRo!OM(q#9v?pQ--^A6rfR7ha+5U)3O?=R|ySva>X*|yi$8(epz>i
zH;2x4Rqn>8cN5o|#}DZpF+=}fmV?9Da*@No5$vq(43<4R2Ka>JMJ#XWsM&6CiNNtU
zK{&Rjg|_di#f0jJnPOy+2I$*~bId&DGGR5?|H8;I>&`LrbDXDbmPb5AFTfZ29=P^}
zqv`h$%S~jGG`BATv8x(yMEC#8bBRuf!{ME8R@-dfaS2XIowG7XA3LvLSrDe~9VI)s
zF5hzL^rkQjDF7Dt(o71{xDI8|fHzie>B!JIKccDfe%wGXW9N+50rT>rWCdwu`UxIU
z8y?=r8sA8eDxnuXqsL4GkN*Sp0nx~}QpfP`>)BpN<4P}07R%rpN1ngQ8@5QD?2)Vr
z3=xuS?#~gdis@t3&ZwLoeryLnOJ_ZxBjIvOQnn_Yw+aSBmZm;ex?DFP4!hbUoR|vR
z!hQSX=AnK)_CNQN7NpD?ji5F|^tnd63uqmMkw4C$XBChGJ{X#c0JdVE^@6p=GG%V
zS{*X&Y*VLf_rR+)RF$rDj2%bXi6*yc#e|0agI-J+MZ-zeTw@D*-ck?f?2==jpU{oW
zz=;z5FfjkM-ODNdUTbE5=EaGuMQ$Z_=eLuD1h-RhL+c;Sb~d6KlGFV|2R`H}-c5(K
zRyUQHEN+Lrqf}eFL@th2SPsNg+58IlgMEQ$hOxy4na?2-BY^6IktNr*R3UHMrbU~-
zV;Le*qj)RNKtJ-(&Lh>O+@T_uMU$t2&y}vmq+M^x4VtF6u0GC@njwg`pFrLpvF+Yx
zX#4eCH&=T^|CE(=-RB&0hH8p(`>{8zE#~cSQ|H$fcP9isD^^X6a3~DAav+jz!N4rb
zgZHo}QkL04w_SFx1_peE4M0c%VN9bu;`-;NZfJ^HSN12`f9Cy1TaAu-X`a?3S;hN*
zk~PZ?n84CWbW1FBQir@1r_8tms(%=f$l7hIS2g+sA?PX%oBI8WfypzN{{^XOElL7B
zrNFX0^~i|)-1xPEKL{wE4Gi}}CJ4QDRB-`qZ2BN>n=hXngJ6qY?V1;*1BH?$mW8hw
zA0&s}w?Trg3WUJoOB?Irm+dbj_4Mn--{iT5`ITq=-rsjpPxR}(4pi6VItJud$Dw%E
z4K_bG6%Pu@;G)?Em~S^)!Z(_bV5VYAi@j~_cs1^Xe$%+tRIp@}q~pxXuL(KGI!@7t
zbop|4X)2Aeq*Idb^X~GoO`1nOGhFmomEhW!Yo+SimZ$9Rm20TgR695t8DE!mK(4jM
z>fjdtfkX$ReZD)g<-ggoO?*0e{_>23jG;G%t0zCB={E2^sBvrqXmnw0*)f3TLK;=j
zR@@&UZ5l?y@^3
zRVDi$rJi&XAv?JFtXE;Oi1Pc_{3Z5(&H2~9wd#`+9HmUPJ+J2dR}m(kH}e~n3m&ka
zBt-BXp0?r&9BYON0|M!geV{oA4Q&{h%o;+=Mh~W|6;%|h@_nmFx#dVu!69j$Cg-$V
z6?4Od&ma<;Ss~GX2*;a(j9fyBTjo*(v^XO{9}$^`5Qx;FD#SC0eP{&(o&Atwmrg80
zW}yA}VAYgG*|1RiozOCwM+0|dc(ck{cFdmbEZTXooNy6*=jiZ^Qa4-{?nw|s4HUFI
zYGlR0gxVCa$4j>qkVZk
zC)|2Qd#`j~-+u38IaaZ}cysK@wZe&VCzYU;<)=3pIk;yRKK)k`*;rd#`-^Ldt^ViU
zjwt-JyCBZ<@8h_Bm34w`LFXglppzQiQuAhv+WZ%;Eh}iMgVj7D)o7{R=IHQXj1L-R?_V3C4Pz!=J3I{rc0MKYuGuH5m86|RK$)8a!E1(ks}_DK@~BSh8-n{s2f
z*{7{ni^?6YVA{lI@NHVtwjcYf?Yr$NHP&9Pmd5ndCY7Vc3R;N6K;0<65B?C{l)rMF
zib+I>-iOQi%2aglBTrEp%-TX;SX6u+3^Fy+^RiT)93l0E3Hf(^tgbJ5ZQ2SKM|!f3
zB4OZishLKN8aHh|0W}3gs+o#&q-Sq_c7S0jY%g{*`?)F2n*
zq4whh)>}qjG5g-VNWA6~Pw|<+vyI=q$6qzJggg2T8+M-YSOxwme=#$M`5asV-iUZ(
zAGU-M(Ri~j-qe7G6Y)Ln4n<3swlovI=yT3>IS6?@d?RzQR%g(wQbUDE{$Vb?J;*%7
z!J=dXz!vaz;-*)xao*zHYa?Z?AQ29>>Q^)-ZDsO8b1KcY(fMU&%=L#W-pz(<7x4Yg
zJHkwy#1Po4=+0gk-8}%2UdJ7tgwbtQ1uPhQmo9NXCv^(bbGm8HdHs@qc`mx@y^31l
zKg^9yFqbVp3Z9l8GR&6pSS3mp8#IC4jDjw?%U)p|qHiYRU|k5YrlwD|K|_znN}u@4
zD*T+=6?BMqU}2ZS95FB&!O&ty{N*_V^l&uFxnAfyH_>?rNYKb^2Wo`N%ivpfXNmbD
znlLLnyk7$3S%)Yr;n%LgFiO^fH*JJ7Uc1l82E^W+V-k-w8LbFj?nrTa^C#E+RKcmx
zvyckQ9z`VZqolc?xBD{jXP&cFp?7{{(E6J&Yq;Iq>Jx7xV=^dv;D>bj9*yw(4h0%_
zJAi(ii9doK0va?`r{M=*;((jqqTjKke^Q-S23buWfud%zX&+}V1nSlT;9Os7;OJ-$a3b^zbJ8YcSn?HAo}kNK0T
zb|Qfzr6)dA9{S6Z5=}&Rr6DVkCh}Wi8^nma)o9FQPNo%l
z
zYkOk`17a$yL_+|s2H>9kqyL%&jUUbYxrR9sy5M_kr?k6mE5l+eZlmNiFFvMNn(hU0(#VJ
zAL`DOv@_akf_E>6%pEMcL*5zFY`@1TD&)y2_yoqy5y&|KQV5VMwp|{39f@$|)S$^&
zl=$cSvG!EcL8?wtG&4Q|=X!ReFt_R5yH?*k_iW*L&Ev{}>bfZp`B7zHn}B8sJG%_*
z!Jc2ppxc+1Jq_!!iUH7(io(kc|Lma#i51`-6sk3#^K$gkJds=ImD|wn{hZ&50^B|0A
zz4>|r0~xM!fiR2@_lFMW;-)bq#vK2r#TX2q@sOG?^GPa6;K?s{wTynyY$Wy-E@p`|Ot
z0$GDIx8US7ux00teS;#TPob$V#x&NIm=54UzN0$>kplFX4nl)b2NqI$8Ba|U7G$3o
zrP0^U(F?otCtIfFBpK5)i~F@!&ko8xbkuvHd3k7!cF1vqz<+m7SRZ%K81eh~k*;eE
zGHa;mYCqLtS)s;zbAOb2yKhxeLJfZu;w8bs^TVRJU2vhlJh$OrmzWcv=^4h9f-nJm
z%N07E7uj|+-}~}Wo3?_!^Qu-;-eP|1n%QMR(|1A#LxOjhNG(@)Xg)qN#i?(??Pi-o
zH80Se4sb8nbfIlpFG2*h-!(GkG!$tH;H5&Ubz1*)V&1fOdPH7?h{x||58j3&xyLcDxg@}6n2mFaEP
zN2h1m?vFCcN@&vn%(0+z(ySqx&K73HgpbxG+-gBY{HSS1*v1Jl~s3^M8sCRf~7f3?WweDy&
zq2-_|P}mp`+;}yo<@px>Zt}Oc)AhM2PvOIRrN1WhIu!>n@8Qv9`(<7V_T0Ix5P0%z
zo~ZVx#U;KZo8i}E7GK7W`aAJA%_Fw2?qwHq3RiJ}kZWAZeEp?yd}8upnnpcEoyLtiCU
zX=vT7?rl2M@|jY&ymle*zIU-np|JAwvNL-0Fp@W0NzEco+C$H*+h@u~z}zX_+}%a|
z#R7lmfCF);=i~?7)<-I8;WY;bmY<#ck9JA_wUxI%$pcddg4s@Gw%QA}IH!iYlWhY@
z(E`P~4PmnP5n_Eb(nH@2ee90f57w3L7Uk!Z8e302ZObItw#M1P3XPr+m3zby&x`jz
zVWWUiR6)3&81{-V6Ai@#Q;Og}KA^iSaQZF@e>!ML | | |