From 8edc6f245b96da63a14c995ccbf870cdb2ab94cb Mon Sep 17 00:00:00 2001 From: WenmuZhou Date: Tue, 20 Oct 2020 17:38:24 +0800 Subject: [PATCH 01/10] =?UTF-8?q?=E5=88=A0=E9=99=A4=E9=85=8D=E7=BD=AE?= =?UTF-8?q?=E6=96=87=E4=BB=B6=E9=87=8C=E7=9A=84=E4=B8=AA=E4=BA=BA=E8=B7=AF?= =?UTF-8?q?=E5=BE=84?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- configs/rec/rec_mv3_none_bilstm_ctc.yml | 12 ++++++------ configs/rec/rec_mv3_none_bilstm_ctc_lmdb.yml | 10 +++++----- configs/rec/rec_r34_vd_none_bilstm_ctc.yml | 4 ++-- 3 files changed, 13 insertions(+), 13 deletions(-) diff --git a/configs/rec/rec_mv3_none_bilstm_ctc.yml b/configs/rec/rec_mv3_none_bilstm_ctc.yml index 7119e0e2bd..3f30afcb4b 100644 --- a/configs/rec/rec_mv3_none_bilstm_ctc.yml +++ b/configs/rec/rec_mv3_none_bilstm_ctc.yml @@ -3,7 +3,7 @@ Global: epoch_num: 500 log_smooth_window: 20 print_batch_step: 10 - save_model_dir: ./output/rec/test/ + save_model_dir: ./output/rec/mv3_none_bilstm_ctc/ save_epoch_step: 500 # evaluation is run every 5000 iterations after the 4000th iteration eval_batch_step: 127 @@ -11,7 +11,7 @@ Global: load_static_weights: True cal_metric_during_train: True pretrained_model: - checkpoints: #output/rec/rec_crnn/best_accuracy + checkpoints: save_inference_dir: use_visualdl: False infer_img: doc/imgs_words/ch/word_1.jpg @@ -66,9 +66,9 @@ Metric: TRAIN: dataset: name: SimpleDataSet - data_dir: /home/zhoujun20/rec + data_dir: ./rec file_list: - - /home/zhoujun20/rec/real_data.txt # dataset1 + - ./rec/real_data.txt # dataset1 ratio_list: [ 0.4,0.6 ] transforms: - DecodeImage: # load image @@ -89,9 +89,9 @@ TRAIN: EVAL: dataset: name: SimpleDataSet - data_dir: /home/zhoujun20/rec + data_dir: ./rec file_list: - - /home/zhoujun20/rec/label_val_all.txt + - ./rec/label_val_all.txt transforms: - DecodeImage: # load image img_mode: BGR diff --git a/configs/rec/rec_mv3_none_bilstm_ctc_lmdb.yml b/configs/rec/rec_mv3_none_bilstm_ctc_lmdb.yml index 1887680ff5..09352d1b06 100644 --- a/configs/rec/rec_mv3_none_bilstm_ctc_lmdb.yml +++ b/configs/rec/rec_mv3_none_bilstm_ctc_lmdb.yml @@ -3,7 +3,7 @@ Global: epoch_num: 500 log_smooth_window: 20 print_batch_step: 1 - save_model_dir: ./output/rec/test/ + save_model_dir: ./output/rec/mv3_none_bilstm_ctc/ save_epoch_step: 500 # evaluation is run every 5000 iterations after the 4000th iteration eval_batch_step: 1016 @@ -11,13 +11,13 @@ Global: load_static_weights: True cal_metric_during_train: True pretrained_model: - checkpoints: #output/rec/rec_crnn/best_accuracy + checkpoints: save_inference_dir: use_visualdl: True infer_img: doc/imgs_words/ch/word_1.jpg # for data or label process max_text_length: 80 - character_dict_path: /home/zhoujun20/rec/lmdb/dict.txt + character_dict_path: ppocr/utils/ppocr_keys_v1.txt character_type: 'ch' use_space_char: True infer_mode: False @@ -67,7 +67,7 @@ TRAIN: dataset: name: LMDBDateSet file_list: - - /home/zhoujun20/rec/lmdb/train # dataset1 + - ./rec/lmdb/train # dataset1 ratio_list: [ 0.4,0.6 ] transforms: - DecodeImage: # load image @@ -89,7 +89,7 @@ EVAL: dataset: name: LMDBDateSet file_list: - - /home/zhoujun20/rec/lmdb/val + - ./rec/lmdb/val transforms: - DecodeImage: # load image img_mode: BGR diff --git a/configs/rec/rec_r34_vd_none_bilstm_ctc.yml b/configs/rec/rec_r34_vd_none_bilstm_ctc.yml index e87115dfc9..adb4195b66 100644 --- a/configs/rec/rec_r34_vd_none_bilstm_ctc.yml +++ b/configs/rec/rec_r34_vd_none_bilstm_ctc.yml @@ -3,7 +3,7 @@ Global: epoch_num: 500 log_smooth_window: 20 print_batch_step: 10 - save_model_dir: ./output/rec/test/ + save_model_dir: ./output/rec/res34_none_bilstm_ctc/ save_epoch_step: 500 # evaluation is run every 5000 iterations after the 4000th iteration eval_batch_step: 127 @@ -11,7 +11,7 @@ Global: load_static_weights: True cal_metric_during_train: True pretrained_model: - checkpoints: #output/rec/rec_crnn/best_accuracy + checkpoints: save_inference_dir: use_visualdl: False infer_img: doc/imgs_words/ch/word_1.jpg From 0968363a89c8c78b6720acb205efa60eeeee9abe Mon Sep 17 00:00:00 2001 From: WenmuZhou Date: Tue, 20 Oct 2020 17:39:07 +0800 Subject: [PATCH 02/10] =?UTF-8?q?=E4=BF=AE=E5=A4=8D=E4=B8=80=E5=A4=84?= =?UTF-8?q?=E6=B3=A8=E9=87=8A=E9=94=99=E8=AF=AF?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- ppocr/modeling/architectures/model.py | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/ppocr/modeling/architectures/model.py b/ppocr/modeling/architectures/model.py index 5723beb7d0..358018d8ab 100644 --- a/ppocr/modeling/architectures/model.py +++ b/ppocr/modeling/architectures/model.py @@ -72,8 +72,7 @@ class Model(nn.Layer): config['Neck']['in_channels'] = in_channels self.neck = build_neck(config['Neck']) in_channels = self.neck.out_channels - - # # build head, head is need for del, rec and cls + # # build head, head is need for det, rec and cls config["Head"]['in_channels'] = in_channels self.head = build_head(config["Head"]) @@ -94,11 +93,11 @@ def check_static(): from ppocr.utils.logging import get_logger from tools import program - config = program.load_config('configs/rec/rec_r34_vd_none_bilstm_ctc.yml') + config = program.load_config('configs/rec/rec_mv3_none_none_ctc_lmdb.yml') logger = get_logger() np.random.seed(0) - data = np.random.rand(1, 3, 32, 320).astype(np.float32) + data = np.random.rand(2, 3, 64, 320).astype(np.float32) paddle.disable_static() config['Architecture']['in_channels'] = 3 @@ -117,7 +116,7 @@ def check_static(): static_out = np.load( '/Users/zhoujun20/Desktop/code/PaddleOCR/output/conv.npy') - diff = y.numpy() - static_out + diff = y.reshape((-1, 6624)).numpy() - static_out print(y.shape, static_out.shape, diff.mean()) From 388d8dae331bca4fe89a15c843af7f7ea3f38079 Mon Sep 17 00:00:00 2001 From: WenmuZhou Date: Tue, 20 Oct 2020 17:47:38 +0800 Subject: [PATCH 03/10] =?UTF-8?q?=E5=88=A0=E9=99=A4=E4=B8=AA=E4=BA=BA?= =?UTF-8?q?=E7=9B=AE=E5=BD=95?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- configs/det/det_mv3_db.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/configs/det/det_mv3_db.yml b/configs/det/det_mv3_db.yml index fc0c007da2..c38efed3ef 100644 --- a/configs/det/det_mv3_db.yml +++ b/configs/det/det_mv3_db.yml @@ -10,8 +10,8 @@ Global: # if pretrained_model is saved in static mode, load_static_weights must set to True load_static_weights: True cal_metric_during_train: False - pretrained_model: /home/zhoujun20/pretrain_models/MobileNetV3_large_x0_5_pretrained - checkpoints: #./output/det_db_0.001_DiceLoss_256_pp_config_2.0b_4gpu/best_accuracy + pretrained_model: ./pretrain_models/MobileNetV3_large_x0_5_pretrained + checkpoints: save_inference_dir: use_visualdl: True infer_img: doc/imgs_en/img_10.jpg From 7c96520de78e19f51056f277686b99f0cd5527f8 Mon Sep 17 00:00:00 2001 From: WenmuZhou Date: Thu, 22 Oct 2020 18:20:20 +0800 Subject: [PATCH 04/10] =?UTF-8?q?yml=E6=96=87=E4=BB=B6=E5=8E=BB=E9=99=A4?= =?UTF-8?q?=E4=B8=AA=E4=BA=BA=E8=B7=AF=E5=BE=84?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- configs/det/det_r50_vd_db.yml | 14 +-- configs/rec/rec_mv3_none_bilstm_ctc.yml | 4 +- configs/rec/rec_mv3_none_none_ctc_lmdb.yml | 30 +++--- configs/rec/rec_r34_vd_none_bilstm_ctc.yml | 8 +- configs/rec/rec_r34_vd_none_none_ctc.yml | 105 +++++++++++++++++++++ 5 files changed, 133 insertions(+), 28 deletions(-) create mode 100644 configs/rec/rec_r34_vd_none_none_ctc.yml diff --git a/configs/det/det_r50_vd_db.yml b/configs/det/det_r50_vd_db.yml index 5794092628..1275c7655c 100644 --- a/configs/det/det_r50_vd_db.yml +++ b/configs/det/det_r50_vd_db.yml @@ -3,15 +3,15 @@ Global: epoch_num: 1200 log_smooth_window: 20 print_batch_step: 2 - save_model_dir: ./output/20201015_r50/ + save_model_dir: ./output/det_r50_vd/ save_epoch_step: 1200 # evaluation is run every 5000 iterations after the 4000th iteration eval_batch_step: 8 # if pretrained_model is saved in static mode, load_static_weights must set to True load_static_weights: True cal_metric_during_train: False - pretrained_model: /home/zhoujun20/pretrain_models/ResNet50_vd_ssld_pretrained/ - checkpoints: #./output/det_db_0.001_DiceLoss_256_pp_config_2.0b_4gpu/best_accuracy + pretrained_model: ./pretrain_models/ResNet50_vd_ssld_pretrained/ + checkpoints: save_inference_dir: use_visualdl: True infer_img: doc/imgs_en/img_10.jpg @@ -65,9 +65,9 @@ Metric: TRAIN: dataset: name: SimpleDataSet - data_dir: /home/zhoujun20/detection/ + data_dir: ./detection/ file_list: - - /home/zhoujun20/detection/train_icdar2015_label.txt # dataset1 + - ./detection/train_icdar2015_label.txt # dataset1 ratio_list: [1.0] transforms: - DecodeImage: # load image @@ -107,9 +107,9 @@ TRAIN: EVAL: dataset: name: SimpleDataSet - data_dir: /home/zhoujun20/detection/ + data_dir: ./detection/ file_list: - - /home/zhoujun20/detection/test_icdar2015_label.txt + - ./detection/test_icdar2015_label.txt transforms: - DecodeImage: # load image img_mode: BGR diff --git a/configs/rec/rec_mv3_none_bilstm_ctc.yml b/configs/rec/rec_mv3_none_bilstm_ctc.yml index 3f30afcb4b..9c45bc70a4 100644 --- a/configs/rec/rec_mv3_none_bilstm_ctc.yml +++ b/configs/rec/rec_mv3_none_bilstm_ctc.yml @@ -68,7 +68,7 @@ TRAIN: name: SimpleDataSet data_dir: ./rec file_list: - - ./rec/real_data.txt # dataset1 + - ./rec/train.txt # dataset1 ratio_list: [ 0.4,0.6 ] transforms: - DecodeImage: # load image @@ -91,7 +91,7 @@ EVAL: name: SimpleDataSet data_dir: ./rec file_list: - - ./rec/label_val_all.txt + - ./rec/val.txt transforms: - DecodeImage: # load image img_mode: BGR diff --git a/configs/rec/rec_mv3_none_none_ctc_lmdb.yml b/configs/rec/rec_mv3_none_none_ctc_lmdb.yml index 413e1c3c31..cc52bf7121 100644 --- a/configs/rec/rec_mv3_none_none_ctc_lmdb.yml +++ b/configs/rec/rec_mv3_none_none_ctc_lmdb.yml @@ -1,25 +1,25 @@ Global: use_gpu: false - epoch_num: 500 + epoch_num: 72 log_smooth_window: 20 - print_batch_step: 1 - save_model_dir: ./output/rec/test/ + print_batch_step: 10 + save_model_dir: ./output/rec/mv3_none_none_ctc/ save_epoch_step: 500 # evaluation is run every 5000 iterations after the 4000th iteration - eval_batch_step: 1016 + eval_batch_step: 2000 # if pretrained_model is saved in static mode, load_static_weights must set to True load_static_weights: True cal_metric_during_train: True pretrained_model: - checkpoints: #output/rec/rec_crnn/best_accuracy + checkpoints: save_inference_dir: use_visualdl: True infer_img: doc/imgs_words/ch/word_1.jpg # for data or label process - max_text_length: 80 - character_dict_path: /home/zhoujun20/rec/lmdb/dict.txt + max_text_length: 25 + character_dict_path: character_type: 'en' - use_space_char: True + use_space_char: False infer_mode: False use_tps: False @@ -29,9 +29,9 @@ Optimizer: beta1: 0.9 beta2: 0.999 learning_rate: - name: Cosine +# name: Cosine lr: 0.0005 - warmup_epoch: 1 +# warmup_epoch: 1 regularizer: name: 'L2' factor: 0.00001 @@ -43,7 +43,7 @@ Architecture: Backbone: name: MobileNetV3 scale: 0.5 - model_name: small + model_name: large small_stride: [ 1, 2, 2, 2 ] Neck: name: SequenceEncoder @@ -66,7 +66,7 @@ TRAIN: dataset: name: LMDBDateSet file_list: - - /Users/zhoujun20/Downloads/evaluation_new # dataset1 + - ./rec/train # dataset1 ratio_list: [ 0.4,0.6 ] transforms: - DecodeImage: # load image @@ -75,7 +75,7 @@ TRAIN: - CTCLabelEncode: # Class handling label - RecAug: - RecResizeImg: - image_shape: [ 3,32,320 ] + image_shape: [ 3,32,100 ] - keepKeys: keep_keys: [ 'image','label','length' ] # dataloader将按照此顺序返回list loader: @@ -88,14 +88,14 @@ EVAL: dataset: name: LMDBDateSet file_list: - - /home/zhoujun20/rec/lmdb/val + - ./rec/val/ transforms: - DecodeImage: # load image img_mode: BGR channel_first: False - CTCLabelEncode: # Class handling label - RecResizeImg: - image_shape: [ 3,32,320 ] + image_shape: [ 3,32,100 ] - keepKeys: keep_keys: [ 'image','label','length' ] # dataloader将按照此顺序返回list loader: diff --git a/configs/rec/rec_r34_vd_none_bilstm_ctc.yml b/configs/rec/rec_r34_vd_none_bilstm_ctc.yml index adb4195b66..06576d315e 100644 --- a/configs/rec/rec_r34_vd_none_bilstm_ctc.yml +++ b/configs/rec/rec_r34_vd_none_bilstm_ctc.yml @@ -64,9 +64,9 @@ Metric: TRAIN: dataset: name: SimpleDataSet - data_dir: /home/zhoujun20/rec + data_dir: ./rec file_list: - - /home/zhoujun20/rec/real_data.txt # dataset1 + - ./rec/train.txt # dataset1 ratio_list: [ 0.4,0.6 ] transforms: - DecodeImage: # load image @@ -87,9 +87,9 @@ TRAIN: EVAL: dataset: name: SimpleDataSet - data_dir: /home/zhoujun20/rec + data_dir: ./rec file_list: - - /home/zhoujun20/rec/label_val_all.txt + - ./rec/val.txt transforms: - DecodeImage: # load image img_mode: BGR diff --git a/configs/rec/rec_r34_vd_none_none_ctc.yml b/configs/rec/rec_r34_vd_none_none_ctc.yml new file mode 100644 index 0000000000..4e2367c966 --- /dev/null +++ b/configs/rec/rec_r34_vd_none_none_ctc.yml @@ -0,0 +1,105 @@ +Global: + use_gpu: false + epoch_num: 500 + log_smooth_window: 20 + print_batch_step: 10 + save_model_dir: ./output/rec/res34_none_none_ctc/ + save_epoch_step: 500 + # evaluation is run every 5000 iterations after the 4000th iteration + eval_batch_step: 127 + # if pretrained_model is saved in static mode, load_static_weights must set to True + load_static_weights: True + cal_metric_during_train: True + pretrained_model: + checkpoints: + save_inference_dir: + use_visualdl: False + infer_img: doc/imgs_words/ch/word_1.jpg + # for data or label process + max_text_length: 80 + character_dict_path: ppocr/utils/ppocr_keys_v1.txt + character_type: 'ch' + use_space_char: False + infer_mode: False + use_tps: False + + +Optimizer: + name: Adam + beta1: 0.9 + beta2: 0.999 + learning_rate: + name: Cosine + lr: 0.001 + warmup_epoch: 4 + regularizer: + name: 'L2' + factor: 0.00001 + +Architecture: + type: rec + algorithm: CRNN + Transform: + Backbone: + name: ResNet + layers: 34 + Neck: + name: SequenceEncoder + encoder_type: reshape + Head: + name: CTC + fc_decay: 0.00001 + +Loss: + name: CTCLoss + +PostProcess: + name: CTCLabelDecode + +Metric: + name: RecMetric + main_indicator: acc + +TRAIN: + dataset: + name: SimpleDataSet + data_dir: ./rec + file_list: + - ./rec/train.txt # dataset1 + ratio_list: [ 0.4,0.6 ] + transforms: + - DecodeImage: # load image + img_mode: BGR + channel_first: False + - CTCLabelEncode: # Class handling label + - RecAug: + - RecResizeImg: + image_shape: [ 3,32,320 ] + - keepKeys: + keep_keys: [ 'image','label','length' ] # dataloader将按照此顺序返回list + loader: + batch_size: 256 + shuffle: True + drop_last: True + num_workers: 8 + +EVAL: + dataset: + name: SimpleDataSet + data_dir: ./rec + file_list: + - ./rec/val.txt + transforms: + - DecodeImage: # load image + img_mode: BGR + channel_first: False + - CTCLabelEncode: # Class handling label + - RecResizeImg: + image_shape: [ 3,32,320 ] + - keepKeys: + keep_keys: [ 'image','label','length' ] # dataloader将按照此顺序返回list + loader: + shuffle: False + drop_last: False + batch_size: 256 + num_workers: 8 From 08b3f98c4f8c449c569d3666beb2fb6c2febec39 Mon Sep 17 00:00:00 2001 From: WenmuZhou Date: Thu, 22 Oct 2020 18:22:56 +0800 Subject: [PATCH 05/10] =?UTF-8?q?=E5=8E=BB=E6=8E=89=E6=B3=A8=E9=87=8A?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- ppocr/modeling/architectures/model.py | 22 ++++++++++------------ 1 file changed, 10 insertions(+), 12 deletions(-) diff --git a/ppocr/modeling/architectures/model.py b/ppocr/modeling/architectures/model.py index 358018d8ab..65769596f7 100644 --- a/ppocr/modeling/architectures/model.py +++ b/ppocr/modeling/architectures/model.py @@ -76,7 +76,6 @@ class Model(nn.Layer): config["Head"]['in_channels'] = in_channels self.head = build_head(config["Head"]) - # @paddle.jit.to_static def forward(self, x): if self.use_transform: x = self.transform(x) @@ -93,30 +92,29 @@ def check_static(): from ppocr.utils.logging import get_logger from tools import program - config = program.load_config('configs/rec/rec_mv3_none_none_ctc_lmdb.yml') + config = program.load_config('configs/det/det_r50_vd_db.yml') logger = get_logger() np.random.seed(0) - data = np.random.rand(2, 3, 64, 320).astype(np.float32) + data = np.random.rand(1, 3, 640, 640).astype(np.float32) + paddle.disable_static() + x = paddle.to_tensor(data) + config['Architecture']['in_channels'] = 3 - config['Architecture']["Head"]['out_channels'] = 6624 + config['Architecture']["Head"]['out_channels'] = 37 model = Model(config['Architecture']) model.eval() load_dygraph_pretrain( - model, - logger, - '/Users/zhoujun20/Desktop/code/PaddleOCR/cnn_ctc/cnn_ctc', - load_static_weights=True) - x = paddle.to_tensor(data) + model, logger, 'det_r50_vd_db/best_accuracy', load_static_weights=True) + y = model(x) for y1 in y: print(y1.shape) - static_out = np.load( - '/Users/zhoujun20/Desktop/code/PaddleOCR/output/conv.npy') - diff = y.reshape((-1, 6624)).numpy() - static_out + static_out = np.load('static_out.npy') + diff = y.numpy() - static_out print(y.shape, static_out.shape, diff.mean()) From 60711bafe2799f05936eb1685108a093903833ca Mon Sep 17 00:00:00 2001 From: WenmuZhou Date: Thu, 22 Oct 2020 18:24:42 +0800 Subject: [PATCH 06/10] =?UTF-8?q?=E6=B7=BB=E5=8A=A0=E9=9D=99=E6=80=81?= =?UTF-8?q?=E5=9B=BE=E7=9A=84create=5Fpredictor?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- tools/infer/utility.py | 56 ++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 56 insertions(+) diff --git a/tools/infer/utility.py b/tools/infer/utility.py index dab06349a7..6e49357b0d 100755 --- a/tools/infer/utility.py +++ b/tools/infer/utility.py @@ -14,11 +14,14 @@ import argparse import os +import sys import cv2 import numpy as np import json from PIL import Image, ImageDraw, ImageFont import math +from paddle.fluid.core import AnalysisConfig +from paddle.fluid.core import create_paddle_predictor def parse_args(): @@ -71,6 +74,59 @@ def parse_args(): return parser.parse_args() +def create_predictor(args, mode, logger): + if mode == "det": + model_dir = args.det_model_dir + elif mode == 'cls': + model_dir = args.cls_model_dir + else: + model_dir = args.rec_model_dir + + if model_dir is None: + logger.info("not find {} model file path {}".format(mode, model_dir)) + sys.exit(0) + model_file_path = model_dir + "/__model__" + params_file_path = model_dir + "/__variables__" + if not os.path.exists(model_file_path): + logger.info("not find model file path {}".format(model_file_path)) + sys.exit(0) + if not os.path.exists(params_file_path): + logger.info("not find params file path {}".format(params_file_path)) + sys.exit(0) + + config = AnalysisConfig(model_file_path, params_file_path) + + if args.use_gpu: + config.enable_use_gpu(args.gpu_mem, 0) + else: + config.disable_gpu() + config.set_cpu_math_library_num_threads(6) + if args.enable_mkldnn: + # cache 10 different shapes for mkldnn to avoid memory leak + config.set_mkldnn_cache_capacity(10) + config.enable_mkldnn() + + # config.enable_memory_optim() + config.disable_glog_info() + + if args.use_zero_copy_run: + config.delete_pass("conv_transpose_eltwiseadd_bn_fuse_pass") + config.switch_use_feed_fetch_ops(False) + else: + config.switch_use_feed_fetch_ops(True) + + predictor = create_paddle_predictor(config) + input_names = predictor.get_input_names() + for name in input_names: + input_tensor = predictor.get_input_tensor(name) + output_names = predictor.get_output_names() + output_tensors = [] + for output_name in output_names: + output_tensor = predictor.get_output_tensor(output_name) + output_tensors.append(output_tensor) + return predictor, input_tensor, output_tensors + + def draw_text_det_res(dt_boxes, img_path): src_im = cv2.imread(img_path) for box in dt_boxes: From 6241b8f9ca0c17933d91646dc0ab4c9827760015 Mon Sep 17 00:00:00 2001 From: WenmuZhou Date: Thu, 22 Oct 2020 19:03:24 +0800 Subject: [PATCH 07/10] =?UTF-8?q?=E6=B7=BB=E5=8A=A0=E9=9D=99=E6=80=81?= =?UTF-8?q?=E5=9B=BE=E7=9A=84create=5Fpredictor?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- tools/infer/utility.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tools/infer/utility.py b/tools/infer/utility.py index 6e49357b0d..cfbec95605 100755 --- a/tools/infer/utility.py +++ b/tools/infer/utility.py @@ -85,8 +85,8 @@ def create_predictor(args, mode, logger): if model_dir is None: logger.info("not find {} model file path {}".format(mode, model_dir)) sys.exit(0) - model_file_path = model_dir + "/__model__" - params_file_path = model_dir + "/__variables__" + model_file_path = model_dir + "/model" + params_file_path = model_dir + "/params" if not os.path.exists(model_file_path): logger.info("not find model file path {}".format(model_file_path)) sys.exit(0) From 122c82e93ff81e32d7f05f82db7e4dcec7456257 Mon Sep 17 00:00:00 2001 From: WenmuZhou Date: Fri, 23 Oct 2020 12:07:44 +0800 Subject: [PATCH 08/10] =?UTF-8?q?=E6=B5=8B=E8=AF=95=E6=A8=A1=E5=BC=8F?= =?UTF-8?q?=E6=97=B6=E5=B0=86=E8=BE=93=E5=87=BAsoftmax=E5=90=8E=E8=BF=94?= =?UTF-8?q?=E5=9B=9E?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- ppocr/modeling/heads/rec_ctc_head.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/ppocr/modeling/heads/rec_ctc_head.py b/ppocr/modeling/heads/rec_ctc_head.py index 8c7b904fed..e96b96ad81 100755 --- a/ppocr/modeling/heads/rec_ctc_head.py +++ b/ppocr/modeling/heads/rec_ctc_head.py @@ -20,6 +20,7 @@ import math import paddle from paddle import ParamAttr, nn +from paddle.nn import functional as F def get_para_bias_attr(l2_decay, k, name): @@ -48,4 +49,6 @@ class CTC(nn.Layer): def forward(self, x, labels=None): predicts = self.fc(x) + if not self.training: + predicts = F.softmax(predicts, axis=2) return predicts From bbe375352e6a6f9fb48d459a0497b5efddb184d9 Mon Sep 17 00:00:00 2001 From: WenmuZhou Date: Tue, 27 Oct 2020 11:23:26 +0800 Subject: [PATCH 09/10] =?UTF-8?q?=E5=88=A0=E9=99=A4check=5Fstatic=E5=87=BD?= =?UTF-8?q?=E6=95=B0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- ppocr/modeling/architectures/model.py | 39 +-------------------------- 1 file changed, 1 insertion(+), 38 deletions(-) diff --git a/ppocr/modeling/architectures/model.py b/ppocr/modeling/architectures/model.py index 65769596f7..222b08d6c8 100644 --- a/ppocr/modeling/architectures/model.py +++ b/ppocr/modeling/architectures/model.py @@ -21,7 +21,6 @@ __dir__ = os.path.dirname(os.path.abspath(__file__)) sys.path.append(__dir__) sys.path.append('/home/zhoujun20/PaddleOCR') -import paddle from paddle import nn from ppocr.modeling.transform import build_transform from ppocr.modeling.backbones import build_backbone @@ -83,40 +82,4 @@ class Model(nn.Layer): if self.use_neck: x = self.neck(x) x = self.head(x) - return x - - -def check_static(): - import numpy as np - from ppocr.utils.save_load import load_dygraph_pretrain - from ppocr.utils.logging import get_logger - from tools import program - - config = program.load_config('configs/det/det_r50_vd_db.yml') - - logger = get_logger() - np.random.seed(0) - data = np.random.rand(1, 3, 640, 640).astype(np.float32) - - paddle.disable_static() - - x = paddle.to_tensor(data) - - config['Architecture']['in_channels'] = 3 - config['Architecture']["Head"]['out_channels'] = 37 - model = Model(config['Architecture']) - model.eval() - load_dygraph_pretrain( - model, logger, 'det_r50_vd_db/best_accuracy', load_static_weights=True) - - y = model(x) - for y1 in y: - print(y1.shape) - - static_out = np.load('static_out.npy') - diff = y.numpy() - static_out - print(y.shape, static_out.shape, diff.mean()) - - -if __name__ == '__main__': - check_static() + return x \ No newline at end of file From e1b39945b8ef67c7ee8cb6d174fcec40ad8e3ab8 Mon Sep 17 00:00:00 2001 From: WenmuZhou Date: Tue, 27 Oct 2020 11:30:23 +0800 Subject: [PATCH 10/10] =?UTF-8?q?=E6=B3=A8=E9=87=8A=E6=9B=B4=E6=94=B9?= =?UTF-8?q?=E4=B8=BA=E8=8B=B1=E6=96=87?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- configs/det/det_mv3_db.yml | 4 +--- configs/det/det_r50_vd_db.yml | 4 +--- configs/rec/rec_mv3_none_bilstm_ctc.yml | 6 ++---- configs/rec/rec_mv3_none_bilstm_ctc_lmdb.yml | 6 ++---- configs/rec/rec_mv3_none_none_ctc_lmdb.yml | 6 ++---- configs/rec/rec_r34_vd_none_bilstm_ctc.yml | 6 ++---- configs/rec/rec_r34_vd_none_none_ctc.yml | 6 ++---- 7 files changed, 12 insertions(+), 26 deletions(-) diff --git a/configs/det/det_mv3_db.yml b/configs/det/det_mv3_db.yml index c38efed3ef..a997aa38fc 100644 --- a/configs/det/det_mv3_db.yml +++ b/configs/det/det_mv3_db.yml @@ -22,9 +22,7 @@ Optimizer: beta1: 0.9 beta2: 0.999 learning_rate: -# name: Cosine lr: 0.001 -# warmup_epoch: 0 regularizer: name: 'L2' factor: 0 @@ -98,7 +96,7 @@ TRAIN: order: 'hwc' - ToCHWImage: - keepKeys: - keep_keys: ['image','threshold_map','threshold_mask','shrink_map','shrink_mask'] # dataloader将按照此顺序返回list + keep_keys: ['image','threshold_map','threshold_mask','shrink_map','shrink_mask'] # dataloader will return list in this order loader: shuffle: True drop_last: False diff --git a/configs/det/det_r50_vd_db.yml b/configs/det/det_r50_vd_db.yml index 1275c7655c..a07273b4ae 100644 --- a/configs/det/det_r50_vd_db.yml +++ b/configs/det/det_r50_vd_db.yml @@ -22,9 +22,7 @@ Optimizer: beta1: 0.9 beta2: 0.999 learning_rate: -# name: Cosine lr: 0.001 -# warmup_epoch: 0 regularizer: name: 'L2' factor: 0 @@ -97,7 +95,7 @@ TRAIN: order: 'hwc' - ToCHWImage: - keepKeys: - keep_keys: ['image','threshold_map','threshold_mask','shrink_map','shrink_mask'] # dataloader将按照此顺序返回list + keep_keys: ['image','threshold_map','threshold_mask','shrink_map','shrink_mask'] # dataloader will return list in this order loader: shuffle: True drop_last: False diff --git a/configs/rec/rec_mv3_none_bilstm_ctc.yml b/configs/rec/rec_mv3_none_bilstm_ctc.yml index 9c45bc70a4..1be7512c9d 100644 --- a/configs/rec/rec_mv3_none_bilstm_ctc.yml +++ b/configs/rec/rec_mv3_none_bilstm_ctc.yml @@ -29,9 +29,7 @@ Optimizer: beta1: 0.9 beta2: 0.999 learning_rate: - name: Cosine lr: 0.001 - warmup_epoch: 4 regularizer: name: 'L2' factor: 0.00001 @@ -79,7 +77,7 @@ TRAIN: - RecResizeImg: image_shape: [ 3,32,320 ] - keepKeys: - keep_keys: [ 'image','label','length' ] # dataloader将按照此顺序返回list + keep_keys: [ 'image','label','length' ] # dataloader will return list in this order loader: batch_size: 256 shuffle: True @@ -100,7 +98,7 @@ EVAL: - RecResizeImg: image_shape: [ 3,32,320 ] - keepKeys: - keep_keys: [ 'image','label','length' ] # dataloader将按照此顺序返回list + keep_keys: [ 'image','label','length' ] # dataloader will return list in this order loader: shuffle: False drop_last: False diff --git a/configs/rec/rec_mv3_none_bilstm_ctc_lmdb.yml b/configs/rec/rec_mv3_none_bilstm_ctc_lmdb.yml index 09352d1b06..f917b0d8ca 100644 --- a/configs/rec/rec_mv3_none_bilstm_ctc_lmdb.yml +++ b/configs/rec/rec_mv3_none_bilstm_ctc_lmdb.yml @@ -29,9 +29,7 @@ Optimizer: beta1: 0.9 beta2: 0.999 learning_rate: - name: Cosine lr: 0.0005 - warmup_epoch: 1 regularizer: name: 'L2' factor: 0.00001 @@ -78,7 +76,7 @@ TRAIN: - RecResizeImg: image_shape: [ 3,32,320 ] - keepKeys: - keep_keys: [ 'image','label','length' ] # dataloader将按照此顺序返回list + keep_keys: [ 'image','label','length' ] # dataloader will return list in this order loader: batch_size: 256 shuffle: True @@ -98,7 +96,7 @@ EVAL: - RecResizeImg: image_shape: [ 3,32,320 ] - keepKeys: - keep_keys: [ 'image','label','length' ] # dataloader将按照此顺序返回list + keep_keys: [ 'image','label','length' ] # dataloader will return list in this order loader: shuffle: False drop_last: False diff --git a/configs/rec/rec_mv3_none_none_ctc_lmdb.yml b/configs/rec/rec_mv3_none_none_ctc_lmdb.yml index cc52bf7121..19997fd599 100644 --- a/configs/rec/rec_mv3_none_none_ctc_lmdb.yml +++ b/configs/rec/rec_mv3_none_none_ctc_lmdb.yml @@ -29,9 +29,7 @@ Optimizer: beta1: 0.9 beta2: 0.999 learning_rate: -# name: Cosine lr: 0.0005 -# warmup_epoch: 1 regularizer: name: 'L2' factor: 0.00001 @@ -77,7 +75,7 @@ TRAIN: - RecResizeImg: image_shape: [ 3,32,100 ] - keepKeys: - keep_keys: [ 'image','label','length' ] # dataloader将按照此顺序返回list + keep_keys: [ 'image','label','length' ] # dataloader will return list in this order loader: batch_size: 256 shuffle: True @@ -97,7 +95,7 @@ EVAL: - RecResizeImg: image_shape: [ 3,32,100 ] - keepKeys: - keep_keys: [ 'image','label','length' ] # dataloader将按照此顺序返回list + keep_keys: [ 'image','label','length' ] # dataloader will return list in this order loader: shuffle: False drop_last: False diff --git a/configs/rec/rec_r34_vd_none_bilstm_ctc.yml b/configs/rec/rec_r34_vd_none_bilstm_ctc.yml index 06576d315e..36e3d1c81c 100644 --- a/configs/rec/rec_r34_vd_none_bilstm_ctc.yml +++ b/configs/rec/rec_r34_vd_none_bilstm_ctc.yml @@ -29,9 +29,7 @@ Optimizer: beta1: 0.9 beta2: 0.999 learning_rate: - name: Cosine lr: 0.001 - warmup_epoch: 4 regularizer: name: 'L2' factor: 0.00001 @@ -77,7 +75,7 @@ TRAIN: - RecResizeImg: image_shape: [ 3,32,320 ] - keepKeys: - keep_keys: [ 'image','label','length' ] # dataloader将按照此顺序返回list + keep_keys: [ 'image','label','length' ] # dataloader will return list in this order loader: batch_size: 256 shuffle: True @@ -98,7 +96,7 @@ EVAL: - RecResizeImg: image_shape: [ 3,32,320 ] - keepKeys: - keep_keys: [ 'image','label','length' ] # dataloader将按照此顺序返回list + keep_keys: [ 'image','label','length' ] # dataloader will return list in this order loader: shuffle: False drop_last: False diff --git a/configs/rec/rec_r34_vd_none_none_ctc.yml b/configs/rec/rec_r34_vd_none_none_ctc.yml index 4e2367c966..641e855b43 100644 --- a/configs/rec/rec_r34_vd_none_none_ctc.yml +++ b/configs/rec/rec_r34_vd_none_none_ctc.yml @@ -29,9 +29,7 @@ Optimizer: beta1: 0.9 beta2: 0.999 learning_rate: - name: Cosine lr: 0.001 - warmup_epoch: 4 regularizer: name: 'L2' factor: 0.00001 @@ -76,7 +74,7 @@ TRAIN: - RecResizeImg: image_shape: [ 3,32,320 ] - keepKeys: - keep_keys: [ 'image','label','length' ] # dataloader将按照此顺序返回list + keep_keys: [ 'image','label','length' ] # dataloader will return list in this order loader: batch_size: 256 shuffle: True @@ -97,7 +95,7 @@ EVAL: - RecResizeImg: image_shape: [ 3,32,320 ] - keepKeys: - keep_keys: [ 'image','label','length' ] # dataloader将按照此顺序返回list + keep_keys: [ 'image','label','length' ] # dataloader will return list in this order loader: shuffle: False drop_last: False