Initial commit

4824c25b · wangsen · 4824c25b · 4824c25b · 4824c25b · 4824c25b
Commit 4824c25b authored Jul 04, 2024 by wangsen
20 changed files
--- a/configs/rec/rec_r34_vd_none_none_ctc.yml
+++ b/configs/rec/rec_r34_vd_none_none_ctc.yml
+Global:
+  use_gpu: true
+  epoch_num: 72
+  log_smooth_window: 20
+  print_batch_step: 10
+  save_model_dir: ./output/rec/r34_vd_none_none_ctc/
+  save_epoch_step: 3
+  # evaluation is run every 2000 iterations
+  eval_batch_step: [0, 2000]
+  cal_metric_during_train: True
+  pretrained_model:
+  checkpoints:
+  save_inference_dir:
+  use_visualdl: False
+  infer_img: doc/imgs_words_en/word_10.png
+  # for data or label process
+  character_dict_path:
+  max_text_length: 25
+  infer_mode: False
+  use_space_char: False
+  save_res_path: ./output/rec/predicts_r34_vd_none_none_ctc.txt
+
+Optimizer:
+  name: Adam
+  beta1: 0.9
+  beta2: 0.999
+  lr:
+    learning_rate: 0.0005
+  regularizer:
+    name: 'L2'
+    factor: 0
+
+Architecture:
+  model_type: rec
+  algorithm: Rosetta
+  Backbone:
+    name: ResNet
+    layers: 34
+  Neck:
+    name: SequenceEncoder
+    encoder_type: reshape
+  Head:
+    name: CTCHead
+    fc_decay: 0.0004
+
+Loss:
+  name: CTCLoss
+
+PostProcess:
+  name: CTCLabelDecode
+
+Metric:
+  name: RecMetric
+  main_indicator: acc
+
+Train:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/training/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - CTCLabelEncode: # Class handling label
+      - RecResizeImg:
+          image_shape: [3, 32, 100]
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'length'] # dataloader will return list in this order
+  loader:
+    shuffle: True
+    batch_size_per_card: 256
+    drop_last: True
+    num_workers: 8
+
+Eval:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/validation/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - CTCLabelEncode: # Class handling label
+      - RecResizeImg:
+          image_shape: [3, 32, 100]
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'length'] # dataloader will return list in this order
+  loader:
+    shuffle: False
+    drop_last: False
+    batch_size_per_card: 256
+    num_workers: 4
--- a/configs/rec/rec_r34_vd_tps_bilstm_att.yml
+++ b/configs/rec/rec_r34_vd_tps_bilstm_att.yml
+Global:
+  use_gpu: True
+  epoch_num: 400
+  log_smooth_window: 20
+  print_batch_step: 10
+  save_model_dir: ./output/rec/b3_rare_r34_none_gru/
+  save_epoch_step: 3
+  # evaluation is run every 5000 iterations after the 4000th iteration
+  eval_batch_step: [0, 2000]
+  cal_metric_during_train: True
+  pretrained_model:
+  checkpoints:
+  save_inference_dir:
+  use_visualdl: False
+  infer_img: doc/imgs_words/ch/word_1.jpg
+  # for data or label process
+  character_dict_path:
+  max_text_length: 25
+  infer_mode: False
+  use_space_char: False
+  save_res_path: ./output/rec/predicts_b3_rare_r34_none_gru.txt
+
+
+Optimizer:
+  name: Adam
+  beta1: 0.9
+  beta2: 0.999
+  lr:
+    learning_rate: 0.0005
+  regularizer:
+    name: 'L2'
+    factor: 0.00000
+
+Architecture:
+  model_type: rec
+  algorithm: RARE
+  Transform:
+    name: TPS
+    num_fiducial: 20
+    loc_lr: 0.1
+    model_name: large
+  Backbone:
+    name: ResNet  
+    layers: 34
+  Neck:
+    name: SequenceEncoder
+    encoder_type: rnn 
+    hidden_size: 256 #96
+  Head:
+    name: AttentionHead  # AttentionHead
+    hidden_size: 256 #
+    l2_decay: 0.00001
+
+Loss:
+  name: AttentionLoss
+
+PostProcess:
+  name: AttnLabelDecode
+
+Metric:
+  name: RecMetric
+  main_indicator: acc
+
+Train:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/training/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - AttnLabelEncode: # Class handling label
+      - RecResizeImg:
+          image_shape: [3, 32, 100]
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'length'] # dataloader will return list in this order
+  loader:
+    shuffle: True
+    batch_size_per_card: 256
+    drop_last: True
+    num_workers: 8
+
+Eval:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/validation/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - AttnLabelEncode: # Class handling label
+      - RecResizeImg:
+          image_shape: [3, 32, 100]
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'length'] # dataloader will return list in this order
+  loader:
+    shuffle: False
+    drop_last: False
+    batch_size_per_card: 256
+    num_workers: 8
--- a/configs/rec/rec_r34_vd_tps_bilstm_ctc.yml
+++ b/configs/rec/rec_r34_vd_tps_bilstm_ctc.yml
+Global:
+  use_gpu: true
+  epoch_num: 72
+  log_smooth_window: 20
+  print_batch_step: 10
+  save_model_dir: ./output/rec/r34_vd_tps_bilstm_ctc/
+  save_epoch_step: 3
+  # evaluation is run every 2000 iterations
+  eval_batch_step: [0, 2000]
+  cal_metric_during_train: True
+  pretrained_model:
+  checkpoints:
+  save_inference_dir:
+  use_visualdl: False
+  infer_img: doc/imgs_words_en/word_10.png
+  # for data or label process
+  character_dict_path:
+  max_text_length: 25
+  infer_mode: False
+  use_space_char: False
+  save_res_path: ./output/rec/predicts_r34_vd_tps_bilstm_ctc.txt
+
+Optimizer:
+  name: Adam
+  beta1: 0.9
+  beta2: 0.999
+  lr:
+    learning_rate: 0.0005
+  regularizer:
+    name: 'L2'
+    factor: 0
+
+Architecture:
+  model_type: rec
+  algorithm: STARNet
+  Transform:
+    name: TPS
+    num_fiducial: 20
+    loc_lr: 0.1
+    model_name: large
+  Backbone:
+    name: ResNet
+    layers: 34
+  Neck:
+    name: SequenceEncoder
+    encoder_type: rnn
+    hidden_size: 256
+  Head:
+    name: CTCHead
+    fc_decay: 0
+
+Loss:
+  name: CTCLoss
+
+PostProcess:
+  name: CTCLabelDecode
+
+Metric:
+  name: RecMetric
+  main_indicator: acc
+
+Train:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/training/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - CTCLabelEncode: # Class handling label
+      - RecResizeImg:
+          image_shape: [3, 32, 100]
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'length'] # dataloader will return list in this order
+  loader:
+    shuffle: True
+    batch_size_per_card: 256
+    drop_last: True
+    num_workers: 8
+
+Eval:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/validation/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - CTCLabelEncode: # Class handling label
+      - RecResizeImg:
+          image_shape: [3, 32, 100]
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'length'] # dataloader will return list in this order
+  loader:
+    shuffle: False
+    drop_last: False
+    batch_size_per_card: 256
+    num_workers: 4
--- a/configs/rec/rec_r45_abinet.yml
+++ b/configs/rec/rec_r45_abinet.yml
+Global:
+  use_gpu: True
+  epoch_num: 10
+  log_smooth_window: 20
+  print_batch_step: 10
+  save_model_dir: ./output/rec/r45_abinet/
+  save_epoch_step: 1
+  # evaluation is run every 2000 iterations
+  eval_batch_step: [0, 2000]
+  cal_metric_during_train: True
+  pretrained_model: ./pretrain_models/abinet_vl_pretrained
+  checkpoints:
+  save_inference_dir:
+  use_visualdl: False
+  infer_img: doc/imgs_words_en/word_10.png
+  # for data or label process
+  character_dict_path:
+  character_type: en
+  max_text_length: 25
+  infer_mode: False
+  use_space_char: False
+  save_res_path: ./output/rec/predicts_abinet.txt
+
+Optimizer:
+  name: Adam
+  beta1: 0.9
+  beta2: 0.99
+  clip_norm: 20.0
+  lr:
+    name: Piecewise
+    decay_epochs: [6]
+    values: [0.0001, 0.00001] 
+  regularizer:
+    name: 'L2'
+    factor: 0.
+
+Architecture:
+  model_type: rec
+  algorithm: ABINet
+  in_channels: 3
+  Transform:
+  Backbone:
+    name: ResNet45
+  Head:
+    name: ABINetHead
+    use_lang: True
+    iter_size: 3
+    
+
+Loss:
+  name: CELoss
+  ignore_index: &ignore_index 100 # Must be greater than the number of character classes
+
+PostProcess:
+  name: ABINetLabelDecode
+
+Metric:
+  name: RecMetric
+  main_indicator: acc
+
+Train:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/training/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: RGB
+          channel_first: False
+      - ABINetRecAug:
+      - ABINetLabelEncode: # Class handling label
+          ignore_index: *ignore_index
+      - ABINetRecResizeImg:
+          image_shape: [3, 32, 128]
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'length'] # dataloader will return list in this order
+  loader:
+    shuffle: True
+    batch_size_per_card: 96
+    drop_last: True
+    num_workers: 4
+
+Eval:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/evaluation/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: RGB
+          channel_first: False
+      - ABINetLabelEncode: # Class handling label
+          ignore_index: *ignore_index
+      - ABINetRecResizeImg:
+          image_shape: [3, 32, 128]
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'length'] # dataloader will return list in this order
+  loader:
+    shuffle: False
+    drop_last: False
+    batch_size_per_card: 256
+    num_workers: 4
+    use_shared_memory: False
--- a/configs/rec/rec_r45_visionlan.yml
+++ b/configs/rec/rec_r45_visionlan.yml
+Global:
+  use_gpu: true
+  epoch_num: 8
+  log_smooth_window: 200
+  print_batch_step: 200
+  save_model_dir: ./output/rec/r45_visionlan
+  save_epoch_step: 1
+  # evaluation is run every 2000 iterations
+  eval_batch_step: [0, 2000]
+  cal_metric_during_train: True
+  pretrained_model:
+  checkpoints: 
+  save_inference_dir:
+  use_visualdl: True
+  infer_img: doc/imgs_words/en/word_2.png
+  # for data or label process
+  character_dict_path:
+  max_text_length: &max_text_length 25
+  training_step: &training_step LA
+  infer_mode: False
+  use_space_char: False
+  save_res_path: ./output/rec/predicts_visionlan.txt
+
+Optimizer:
+  name: Adam
+  beta1: 0.9
+  beta2: 0.999
+  clip_norm: 20.0
+  group_lr: true
+  training_step: *training_step
+  lr:
+    name: Piecewise
+    decay_epochs: [6]
+    values: [0.0001, 0.00001] 
+  regularizer:
+    name: 'L2'
+    factor: 0
+
+Architecture:
+  model_type: rec
+  algorithm: VisionLAN
+  Transform:
+  Backbone:
+    name: ResNet45
+    strides: [2, 2, 2, 1, 1]
+  Head:
+    name: VLHead
+    n_layers: 3
+    n_position: 256
+    n_dim: 512
+    max_text_length: *max_text_length
+    training_step: *training_step
+
+Loss:
+  name: VLLoss
+  mode: *training_step
+  weight_res: 0.5
+  weight_mas: 0.5
+
+PostProcess:
+  name: VLLabelDecode
+
+Metric:
+  name: RecMetric
+  is_filter: true
+
+
+Train:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/training/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: RGB
+          channel_first: False
+      - ABINetRecAug:
+      - VLLabelEncode: # Class handling label
+      - VLRecResizeImg:
+          image_shape: [3, 64, 256]
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'label_res', 'label_sub', 'label_id', 'length'] # dataloader will return list in this order
+  loader:
+    shuffle: True
+    batch_size_per_card: 220
+    drop_last: True
+    num_workers: 4
+
+Eval:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/validation/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: RGB
+          channel_first: False
+      - VLLabelEncode: # Class handling label
+      - VLRecResizeImg:
+          image_shape: [3, 64, 256]
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'label_res', 'label_sub', 'label_id', 'length'] # dataloader will return list in this order
+  loader:
+    shuffle: False
+    drop_last: False
+    batch_size_per_card: 64
+    num_workers: 4
+  
--- a/configs/rec/rec_r50_fpn_srn.yml
+++ b/configs/rec/rec_r50_fpn_srn.yml
+Global:
+  use_gpu: True
+  epoch_num: 72
+  log_smooth_window: 20
+  print_batch_step: 5
+  save_model_dir: ./output/rec/srn_new
+  save_epoch_step: 3
+  # evaluation is run every 5000 iterations after the 4000th iteration
+  eval_batch_step: [0, 5000]
+  cal_metric_during_train: True
+  pretrained_model: 
+  checkpoints:
+  save_inference_dir:
+  use_visualdl: False
+  infer_img: doc/imgs_words/ch/word_1.jpg
+  # for data or label process
+  character_dict_path:
+  max_text_length: 25
+  num_heads: 8
+  infer_mode: False
+  use_space_char: False
+  save_res_path: ./output/rec/predicts_srn.txt
+
+
+Optimizer:
+  name: Adam
+  beta1: 0.9
+  beta2: 0.999
+  clip_norm: 10.0
+  lr:
+    learning_rate: 0.0001
+
+Architecture:
+  model_type: rec
+  algorithm: SRN
+  in_channels: 1
+  Transform:
+  Backbone:
+    name: ResNetFPN
+  Head:
+    name: SRNHead
+    max_text_length: 25
+    num_heads: 8
+    num_encoder_TUs: 2
+    num_decoder_TUs: 4
+    hidden_dims: 512
+
+Loss:
+  name: SRNLoss
+
+PostProcess:
+  name: SRNLabelDecode
+
+Metric:
+  name: RecMetric
+  main_indicator: acc
+
+Train:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/training/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - SRNLabelEncode: # Class handling label
+      - SRNRecResizeImg:
+          image_shape: [1, 64, 256]
+      - KeepKeys:
+          keep_keys: ['image',
+                      'label',
+                      'length',
+                      'encoder_word_pos',
+                      'gsrm_word_pos',
+                      'gsrm_slf_attn_bias1',
+                      'gsrm_slf_attn_bias2'] # dataloader will return list in this order
+  loader:
+    shuffle: False
+    batch_size_per_card: 64
+    drop_last: False
+    num_workers: 4
+
+Eval:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/validation/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - SRNLabelEncode: # Class handling label
+      - SRNRecResizeImg:
+          image_shape: [1, 64, 256]
+      - KeepKeys:
+          keep_keys: ['image',
+                      'label',
+                      'length',
+                      'encoder_word_pos',
+                      'gsrm_word_pos',
+                      'gsrm_slf_attn_bias1',
+                      'gsrm_slf_attn_bias2'] 
+  loader:
+    shuffle: False
+    drop_last: False
+    batch_size_per_card: 32
+    num_workers: 4
--- a/configs/rec/rec_resnet_rfl_att.yml
+++ b/configs/rec/rec_resnet_rfl_att.yml
+Global:
+  use_gpu: True
+  epoch_num: 6
+  log_smooth_window: 20
+  print_batch_step: 50
+  save_model_dir: ./output/rec/rec_resnet_rfl_att/
+  save_epoch_step: 1
+  # evaluation is run every 5000 iterations after the 4000th iteration
+  eval_batch_step: [0, 5000]
+  cal_metric_during_train: True
+  pretrained_model: ./pretrain_models/rec_resnet_rfl_visual/best_accuracy.pdparams 
+  checkpoints:
+  save_inference_dir:
+  use_visualdl: False
+  infer_img: doc/imgs_words_en/word_10.png
+  # for data or label process
+  character_dict_path:
+  max_text_length: 25
+  infer_mode: False
+  use_space_char: False
+  save_res_path: ./output/rec/rec_resnet_rfl.txt
+
+
+Optimizer:
+  name: AdamW
+  beta1: 0.9
+  beta2: 0.999
+  weight_decay: 0.0
+  clip_norm_global: 5.0
+  lr:
+    name: Piecewise
+    decay_epochs : [3, 4, 5]
+    values : [0.001, 0.0003, 0.00009, 0.000027]
+
+Architecture:
+  model_type: rec
+  algorithm: RFL
+  in_channels: 1
+  Transform:
+    name: TPS
+    num_fiducial: 20
+    loc_lr: 1.0
+    model_name: large
+  Backbone:
+    name: ResNetRFL
+    use_cnt: True
+    use_seq: True
+  Neck:
+    name: RFAdaptor
+    use_v2s: True
+    use_s2v: True
+  Head:
+    name: RFLHead  
+    in_channels: 512
+    hidden_size: 256
+    batch_max_legnth: 25
+    out_channels: 38
+    use_cnt: True
+    use_seq: True
+
+Loss:
+  name: RFLLoss
+  # ignore_index: 0
+
+PostProcess:
+  name: RFLLabelDecode
+
+Metric:
+  name: RecMetric
+  main_indicator: acc
+
+Train:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/training
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - RFLLabelEncode: # Class handling label
+      - RFLRecResizeImg:
+          image_shape: [1, 32, 100]
+          padding: false
+          interpolation: 2
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'length', 'cnt_label'] # dataloader will return list in this order
+  loader:
+    shuffle: True
+    batch_size_per_card: 64
+    drop_last: True
+    num_workers: 8
+
+Eval:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/validation/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - RFLLabelEncode: # Class handling label
+      - RFLRecResizeImg:
+          image_shape: [1, 32, 100]
+          padding: false
+          interpolation: 2
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'length', 'cnt_label'] # dataloader will return list in this order
+  loader:
+    shuffle: False
+    drop_last: False
+    batch_size_per_card: 256
+    num_workers: 8
--- a/configs/rec/rec_resnet_rfl_visual.yml
+++ b/configs/rec/rec_resnet_rfl_visual.yml
+Global:
+  use_gpu: True
+  epoch_num: 6
+  log_smooth_window: 20
+  print_batch_step: 50
+  save_model_dir: ./output/rec/rec_resnet_rfl_visual/
+  save_epoch_step: 1
+  # evaluation is run every 5000 iterations after the 4000th iteration
+  eval_batch_step: [0, 5000]
+  cal_metric_during_train: False
+  pretrained_model:
+  checkpoints: 
+  save_inference_dir:
+  use_visualdl: False
+  infer_img: doc/imgs_words_en/word_10.png
+  # for data or label process
+  character_dict_path:
+  max_text_length: 25
+  infer_mode: False
+  use_space_char: False
+  save_res_path: ./output/rec/rec_resnet_rfl_visual.txt
+
+
+Optimizer:
+  name: AdamW
+  beta1: 0.9
+  beta2: 0.999
+  weight_decay: 0.0
+  clip_norm_global: 5.0
+  lr:
+    name: Piecewise
+    decay_epochs : [3, 4, 5]
+    values : [0.001, 0.0003, 0.00009, 0.000027]
+
+Architecture:
+  model_type: rec
+  algorithm: RFL
+  in_channels: 1
+  Transform:
+    name: TPS
+    num_fiducial: 20
+    loc_lr: 1.0
+    model_name: large
+  Backbone:
+    name: ResNetRFL
+    use_cnt: True
+    use_seq: False
+  Neck:
+    name: RFAdaptor
+    use_v2s: False
+    use_s2v: False
+  Head:
+    name: RFLHead  
+    in_channels: 512
+    hidden_size: 256
+    batch_max_legnth: 25
+    out_channels: 38
+    use_cnt: True
+    use_seq: False
+Loss:
+  name: RFLLoss
+
+PostProcess:
+  name: RFLLabelDecode
+
+Metric:
+  name: CNTMetric
+  main_indicator: acc
+
+Train:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/training
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - RFLLabelEncode: # Class handling label
+      - RFLRecResizeImg:
+          image_shape: [1, 32, 100]
+          padding: false
+          interpolation: 2
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'length', 'cnt_label'] # dataloader will return list in this order
+  loader:
+    shuffle: True
+    batch_size_per_card: 64
+    drop_last: True
+    num_workers: 8
+
+Eval:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/evaluation
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - RFLLabelEncode: # Class handling label
+      - RFLRecResizeImg:
+          image_shape: [1, 32, 100]
+          padding: false
+          interpolation: 2
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'length', 'cnt_label'] # dataloader will return list in this order
+  loader:
+    shuffle: False
+    drop_last: False
+    batch_size_per_card: 256
+    num_workers: 8
--- a/configs/rec/rec_resnet_stn_bilstm_att.yml
+++ b/configs/rec/rec_resnet_stn_bilstm_att.yml
+Global:
+  use_gpu: True
+  epoch_num: 6
+  log_smooth_window: 20
+  print_batch_step: 10
+  save_model_dir: ./output/rec/seed
+  save_epoch_step: 3
+  # evaluation is run every 5000 iterations after the 4000th iteration
+  eval_batch_step: [0, 2000]
+  cal_metric_during_train: True
+  pretrained_model:
+  checkpoints:
+  save_inference_dir:
+  use_visualdl: False
+  infer_img: doc/imgs_words_en/word_10.png
+  # for data or label process
+  character_dict_path: ppocr/utils/EN_symbol_dict.txt
+  max_text_length: 100
+  infer_mode: False
+  use_space_char: False
+  save_res_path: ./output/rec/predicts_seed.txt
+
+
+Optimizer:
+  name: Adadelta
+  weight_deacy: 0.0
+  momentum: 0.9
+  lr:
+    name: Piecewise
+    decay_epochs: [4, 5]
+    values: [1.0, 0.1, 0.01]
+  regularizer:
+    name: 'L2'
+    factor: 2.0e-05
+
+
+Architecture:
+  model_type: rec
+  algorithm: SEED
+  Transform:
+    name: STN_ON
+    tps_inputsize: [32, 64]
+    tps_outputsize: [32, 100]
+    num_control_points: 20
+    tps_margins: [0.05,0.05]
+    stn_activation: none
+  Backbone:
+    name: ResNet_ASTER
+  Head:
+    name: AsterHead  # AttentionHead
+    sDim: 512
+    attDim: 512
+    max_len_labels: 100
+
+Loss:
+  name: AsterLoss
+
+PostProcess:
+  name: SEEDLabelDecode
+
+Metric:
+  name: RecMetric
+  main_indicator: acc
+  is_filter: True
+
+Train:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/training/
+    transforms:
+      - Fasttext:
+          path: "./cc.en.300.bin"
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - SEEDLabelEncode: # Class handling label
+      - RecResizeImg:
+          character_dict_path:
+          image_shape: [3, 64, 256]
+          padding: False
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'length', 'fast_label'] # dataloader will return list in this order
+  loader:
+    shuffle: True
+    batch_size_per_card: 256
+    drop_last: True
+    num_workers: 6
+
+Eval:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/evaluation/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - SEEDLabelEncode: # Class handling label
+      - RecResizeImg:
+          character_dict_path:
+          image_shape: [3, 64, 256]
+          padding: False
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'length'] # dataloader will return list in this order
+  loader:
+    shuffle: False
+    drop_last: True
+    batch_size_per_card: 256
+    num_workers: 4
--- a/configs/rec/rec_satrn.yml
+++ b/configs/rec/rec_satrn.yml
+Global:
+  use_gpu: true
+  epoch_num: 5
+  log_smooth_window: 20
+  print_batch_step: 50
+  save_model_dir: ./output/rec/rec_satrn/
+  save_epoch_step: 1
+  # evaluation is run every 5000 iterations
+  eval_batch_step: [0, 5000]
+  cal_metric_during_train: False
+  pretrained_model:
+  checkpoints:
+  save_inference_dir:
+  use_visualdl: False
+  infer_img: 
+  # for data or label process
+  character_dict_path: ppocr/utils/dict90.txt
+  max_text_length: 25
+  infer_mode: False
+  use_space_char: False
+  rm_symbol: True
+  save_res_path: ./output/rec/predicts_satrn.txt
+
+Optimizer:
+  name: Adam
+  beta1: 0.9
+  beta2: 0.999
+  lr:
+    name: Piecewise
+    decay_epochs: [3, 4]
+    values: [0.0003, 0.00003, 0.000003] 
+  regularizer:
+    name: 'L2'
+    factor: 0
+
+Architecture:
+  model_type: rec
+  algorithm: SATRN
+  Backbone:
+    name: ShallowCNN
+    in_channels: 3
+    hidden_dim: 256
+  Head:
+    name: SATRNHead
+    enc_cfg:
+      n_layers: 6
+      n_head: 8
+      d_k: 32
+      d_v: 32
+      d_model: 256
+      n_position: 100
+      d_inner: 1024
+      dropout: 0.1
+    dec_cfg:
+      n_layers: 6
+      d_embedding: 256
+      n_head: 8
+      d_model: 256
+      d_inner: 1024
+      d_k: 32
+      d_v: 32
+      max_seq_len: 25
+      start_idx: 91
+
+Loss:
+  name: SATRNLoss
+
+PostProcess:
+  name: SATRNLabelDecode
+
+Metric:
+  name: RecMetric
+  main_indicator: acc
+
+Train:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/training/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - SATRNLabelEncode: # Class handling label
+      - SVTRRecResizeImg:
+          image_shape: [3, 32, 100] 
+          padding: False
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'valid_ratio'] # dataloader will return list in this order
+  loader:
+    shuffle: True
+    batch_size_per_card: 128
+    drop_last: True
+    num_workers: 8
+    use_shared_memory: False
+
+Eval:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/evaluation/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - SATRNLabelEncode: # Class handling label
+      - SVTRRecResizeImg:
+          image_shape: [3, 32, 100] 
+          padding: False
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'valid_ratio'] # dataloader will return list in this order
+  
+  loader:
+    shuffle: False
+    drop_last: False
+    batch_size_per_card: 128 
+    num_workers: 4
+    use_shared_memory: False
+
--- a/configs/rec/rec_svtrnet.yml
+++ b/configs/rec/rec_svtrnet.yml
+Global:
+  use_gpu: True
+  epoch_num: 20
+  log_smooth_window: 20
+  print_batch_step: 10
+  save_model_dir: ./output/rec/svtr/
+  save_epoch_step: 1
+  # evaluation is run every 2000 iterations after the 0th iteration
+  eval_batch_step: [0, 2000]
+  cal_metric_during_train: True
+  pretrained_model:
+  checkpoints:
+  save_inference_dir:
+  use_visualdl: False
+  infer_img: doc/imgs_words_en/word_10.png
+  # for data or label process
+  character_dict_path:
+  character_type: en
+  max_text_length: 25
+  infer_mode: False
+  use_space_char: False
+  save_res_path: ./output/rec/predicts_svtr_tiny.txt
+  d2s_train_image_shape: [3, 64, 256]
+
+
+Optimizer:
+  name: AdamW
+  beta1: 0.9
+  beta2: 0.99
+  epsilon: 1.e-8
+  weight_decay: 0.05
+  no_weight_decay_name: norm pos_embed
+  one_dim_param_no_weight_decay: True
+  lr:
+    name: Cosine
+    learning_rate: 0.0005
+    warmup_epoch: 2
+
+Architecture:
+  model_type: rec
+  algorithm: SVTR
+  Transform:
+    name: STN_ON
+    tps_inputsize: [32, 64]
+    tps_outputsize: [32, 100]
+    num_control_points: 20
+    tps_margins: [0.05,0.05]
+    stn_activation: none
+  Backbone:
+    name: SVTRNet
+    img_size: [32, 100]
+    out_char_num: 25 # W//4 or W//8 or W/12
+    out_channels: 192
+    patch_merging: 'Conv'
+    embed_dim: [64, 128, 256]
+    depth: [3, 6, 3]
+    num_heads: [2, 4, 8]
+    mixer: ['Local','Local','Local','Local','Local','Local','Global','Global','Global','Global','Global','Global']
+    local_mixer: [[7, 11], [7, 11], [7, 11]]
+    last_stage: True
+    prenorm: False
+  Neck:
+    name: SequenceEncoder
+    encoder_type: reshape
+  Head:
+    name: CTCHead
+
+Loss:
+  name: CTCLoss
+
+PostProcess:
+  name: CTCLabelDecode
+
+Metric:
+  name: RecMetric
+  main_indicator: acc
+
+Train:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/training/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - SVTRRecAug:
+          aug_type: 0 # or 1
+      - CTCLabelEncode: # Class handling label
+      - SVTRRecResizeImg:
+          image_shape: [3, 64, 256]
+          padding: False
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'length'] # dataloader will return list in this order
+  loader:
+    shuffle: True
+    batch_size_per_card: 512
+    drop_last: True
+    num_workers: 8
+
+Eval:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/evaluation/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - CTCLabelEncode: # Class handling label
+      - SVTRRecResizeImg:
+          image_shape: [3, 64, 256]
+          padding: False
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'length'] # dataloader will return list in this order
+  loader:
+    shuffle: False
+    drop_last: False
+    batch_size_per_card: 256
+    num_workers: 2
--- a/configs/rec/rec_svtrnet_ch.yml
+++ b/configs/rec/rec_svtrnet_ch.yml
+Global:
+  use_gpu: true
+  epoch_num: 100
+  log_smooth_window: 20
+  print_batch_step: 10
+  save_model_dir: ./output/rec/svtr_ch_all/
+  save_epoch_step: 10
+  eval_batch_step:
+  - 0
+  - 2000
+  cal_metric_during_train: true
+  pretrained_model: null
+  checkpoints: null
+  save_inference_dir: null
+  use_visualdl: false
+  infer_img: doc/imgs_words/ch/word_1.jpg
+  character_dict_path: ppocr/utils/ppocr_keys_v1.txt
+  max_text_length: 25
+  infer_mode: false
+  use_space_char: true
+  save_res_path: ./output/rec/predicts_svtr_tiny_ch_all.txt
+  d2s_train_image_shape: [3, 32, 320]
+Optimizer:
+  name: AdamW
+  beta1: 0.9
+  beta2: 0.99
+  epsilon: 1.0e-08
+  weight_decay: 0.05
+  no_weight_decay_name: norm pos_embed
+  one_dim_param_no_weight_decay: true
+  lr:
+    name: Cosine
+    learning_rate: 0.0005
+    warmup_epoch: 2
+Architecture:
+  model_type: rec
+  algorithm: SVTR
+  Transform: null
+  Backbone:
+    name: SVTRNet
+    img_size:
+    - 32
+    - 320
+    out_char_num: 40 # W//4 or W//8 or W/12
+    out_channels: 96
+    patch_merging: Conv
+    embed_dim:
+    - 64
+    - 128
+    - 256
+    depth:
+    - 3
+    - 6
+    - 3
+    num_heads:
+    - 2
+    - 4
+    - 8
+    mixer:
+    - Local
+    - Local
+    - Local
+    - Local
+    - Local
+    - Local
+    - Global
+    - Global
+    - Global
+    - Global
+    - Global
+    - Global
+    local_mixer:
+    - - 7
+      - 11
+    - - 7
+      - 11
+    - - 7
+      - 11
+    last_stage: true
+    prenorm: false
+  Neck:
+    name: SequenceEncoder
+    encoder_type: reshape
+  Head:
+    name: CTCHead
+Loss:
+  name: CTCLoss
+PostProcess:
+  name: CTCLabelDecode
+Metric:
+  name: RecMetric
+  main_indicator: acc
+Train:
+  dataset:
+    name: SimpleDataSet
+    data_dir: ./train_data
+    label_file_list:
+    - ./train_data/train_list.txt
+    ext_op_transform_idx: 1
+    transforms:
+    - DecodeImage:
+        img_mode: BGR
+        channel_first: false
+    - RecConAug:
+        prob: 0.5
+        ext_data_num: 2
+        image_shape:
+        - 32
+        - 320
+        - 3
+    - RecAug: null
+    - CTCLabelEncode: null
+    - SVTRRecResizeImg:
+        image_shape:
+        - 3
+        - 32
+        - 320
+        padding: true
+    - KeepKeys:
+        keep_keys:
+        - image
+        - label
+        - length
+  loader:
+    shuffle: true
+    batch_size_per_card: 256
+    drop_last: true
+    num_workers: 8
+Eval:
+  dataset:
+    name: SimpleDataSet
+    data_dir: ./train_data
+    label_file_list:
+    - ./train_data/val_list.txt
+    transforms:
+    - DecodeImage:
+        img_mode: BGR
+        channel_first: false
+    - CTCLabelEncode: null
+    - SVTRRecResizeImg:
+        image_shape:
+        - 3
+        - 32
+        - 320
+        padding: true
+    - KeepKeys:
+        keep_keys:
+        - image
+        - label
+        - length
+  loader:
+    shuffle: false
+    drop_last: false
+    batch_size_per_card: 256
+    num_workers: 2
+profiler_options: null
--- a/configs/rec/rec_vitstr_none_ce.yml
+++ b/configs/rec/rec_vitstr_none_ce.yml
+Global:
+  use_gpu: True
+  epoch_num: 20
+  log_smooth_window: 20
+  print_batch_step: 10
+  save_model_dir: ./output/rec/vitstr_none_ce/
+  save_epoch_step: 1
+  # evaluation is run every 2000 iterations after the 0th iteration#
+  eval_batch_step: [0, 2000]
+  cal_metric_during_train: True
+  pretrained_model:
+  checkpoints:
+  save_inference_dir:
+  use_visualdl: False
+  infer_img: doc/imgs_words_en/word_10.png
+  # for data or label process
+  character_dict_path: ppocr/utils/EN_symbol_dict.txt
+  max_text_length: 25
+  infer_mode: False
+  use_space_char: False
+  save_res_path: ./output/rec/predicts_vitstr.txt
+
+
+Optimizer:
+  name: Adadelta
+  epsilon: 1.e-8
+  rho: 0.95
+  clip_norm: 5.0
+  lr:
+    learning_rate: 1.0
+
+Architecture:
+  model_type: rec
+  algorithm: ViTSTR
+  in_channels: 1
+  Transform:
+  Backbone:
+    name: ViTSTR
+    scale: tiny
+  Neck:
+    name: SequenceEncoder
+    encoder_type: reshape
+  Head:
+    name: CTCHead
+
+Loss:
+  name: CELoss
+  with_all: True
+  ignore_index: &ignore_index 0 # Must be zero or greater than the number of character classes
+
+PostProcess:
+  name: ViTSTRLabelDecode
+
+Metric:
+  name: RecMetric
+  main_indicator: acc
+
+Train:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/training/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - ViTSTRLabelEncode: # Class handling label
+          ignore_index: *ignore_index
+      - GrayRecResizeImg:
+          image_shape: [224, 224] # W H
+          resize_type: PIL # PIL or OpenCV
+          inter_type: 'Image.BICUBIC'
+          scale: false
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'length'] # dataloader will return list in this order
+  loader:
+    shuffle: True
+    batch_size_per_card: 48
+    drop_last: True
+    num_workers: 8
+
+Eval:
+  dataset:
+    name: LMDBDataSet
+    data_dir: ./train_data/data_lmdb_release/evaluation/
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - ViTSTRLabelEncode: # Class handling label
+          ignore_index: *ignore_index
+      - GrayRecResizeImg:
+          image_shape: [224, 224] # W H
+          resize_type: PIL # PIL or OpenCV
+          inter_type: 'Image.BICUBIC'
+          scale: false
+      - KeepKeys:
+          keep_keys: ['image', 'label', 'length'] # dataloader will return list in this order
+  loader:
+    shuffle: False
+    drop_last: False
+    batch_size_per_card: 256
+    num_workers: 2
--- a/configs/sr/sr_telescope.yml
+++ b/configs/sr/sr_telescope.yml
+Global:
+  use_gpu: true
+  epoch_num: 100
+  log_smooth_window: 20
+  print_batch_step: 10
+  save_model_dir: ./output/sr/sr_telescope/
+  save_epoch_step: 3
+  # evaluation is run every 2000 iterations
+  eval_batch_step: [0, 1000]
+  cal_metric_during_train: False
+  pretrained_model:
+  checkpoints:
+  save_inference_dir:  ./output/sr/sr_telescope/infer
+  use_visualdl: False
+  infer_img: doc/imgs_words_en/word_52.png
+  # for data or label process
+  character_dict_path:
+  max_text_length: 100
+  infer_mode: False
+  use_space_char: False
+  save_res_path: ./output/sr/predicts_telescope.txt
+
+Optimizer:
+  name: Adam
+  beta1: 0.5
+  beta2: 0.999
+  clip_norm: 0.25
+  lr:
+    learning_rate: 0.0001
+
+Architecture:
+  model_type: sr
+  algorithm: Telescope
+  Transform:
+    name: TBSRN
+    STN: True
+    infer_mode: False
+
+Loss:
+  name: TelescopeLoss
+  confuse_dict_path: ./ppocr/utils/dict/confuse.pkl
+
+
+PostProcess:
+  name: None
+
+Metric:
+  name: SRMetric
+  main_indicator: all
+
+Train:
+  dataset:
+    name: LMDBDataSetSR
+    data_dir: ./train_data/TextZoom/train
+    transforms:
+      - SRResize:
+          imgH: 32
+          imgW: 128
+          down_sample_scale: 2
+      - KeepKeys:
+          keep_keys: ['img_lr', 'img_hr', 'label'] # dataloader will return list in this order
+  loader:
+    shuffle: False
+    batch_size_per_card: 16
+    drop_last: True
+    num_workers: 4
+
+Eval:
+  dataset:
+    name: LMDBDataSetSR
+    data_dir: ./train_data/TextZoom/test
+    transforms:
+      - SRResize:
+          imgH: 32
+          imgW: 128
+          down_sample_scale: 2
+      - KeepKeys:
+          keep_keys: ['img_lr', 'img_hr', 'label'] # dataloader will return list in this order
+  loader:
+    shuffle: False
+    drop_last: False
+    batch_size_per_card: 16
+    num_workers: 4
+
--- a/configs/sr/sr_tsrn_transformer_strock.yml
+++ b/configs/sr/sr_tsrn_transformer_strock.yml
+Global:
+  use_gpu: true
+  epoch_num: 500
+  log_smooth_window: 20
+  print_batch_step: 10
+  save_model_dir: ./output/sr/sr_tsrn_transformer_strock/
+  save_epoch_step: 3
+  # evaluation is run every 2000 iterations
+  eval_batch_step: [0, 1000]
+  cal_metric_during_train: False
+  pretrained_model:
+  checkpoints:
+  save_inference_dir: sr_output
+  use_visualdl: False
+  infer_img: doc/imgs_words_en/word_52.png
+  # for data or label process
+  character_dict_path: ./train_data/srdata/english_decomposition.txt
+  max_text_length: 100
+  infer_mode: False
+  use_space_char: False
+  save_res_path: ./output/sr/predicts_gestalt.txt
+
+Optimizer:
+  name: Adam
+  beta1: 0.5
+  beta2: 0.999
+  clip_norm: 0.25
+  lr:
+    learning_rate: 0.0001
+
+Architecture:
+  model_type: sr
+  algorithm: Gestalt
+  Transform:
+    name: TSRN
+    STN: True
+    infer_mode: False
+
+Loss:
+  name: StrokeFocusLoss
+  character_dict_path: ./train_data/srdata/english_decomposition.txt
+
+PostProcess:
+  name: None
+
+Metric:
+  name: SRMetric
+  main_indicator: all
+
+Train:
+  dataset:
+    name: LMDBDataSetSR
+    data_dir: ./train_data/srdata/train
+    transforms:
+      - SRResize:
+          imgH: 32
+          imgW: 128
+          down_sample_scale: 2
+      - SRLabelEncode: # Class handling label
+      - KeepKeys:
+          keep_keys: ['img_lr', 'img_hr', 'length', 'input_tensor', 'label'] # dataloader will return list in this order
+  loader:
+    shuffle: False
+    batch_size_per_card: 16
+    drop_last: True
+    num_workers: 4
+
+Eval:
+  dataset:
+    name: LMDBDataSetSR
+    data_dir: ./train_data/srdata/test
+    transforms:
+      - SRResize:
+          imgH: 32
+          imgW: 128
+          down_sample_scale: 2
+      - SRLabelEncode: # Class handling label
+      - KeepKeys:
+          keep_keys: ['img_lr', 'img_hr','length', 'input_tensor', 'label'] # dataloader will return list in this order
+  loader:
+    shuffle: False
+    drop_last: False
+    batch_size_per_card: 16
+    num_workers: 4
+
--- a/configs/table/SLANet.yml
+++ b/configs/table/SLANet.yml
+Global:
+  use_gpu: true
+  epoch_num: 100
+  log_smooth_window: 20
+  print_batch_step: 20
+  save_model_dir: ./output/SLANet
+  save_epoch_step: 400
+  # evaluation is run every 1000 iterations after the 0th iteration
+  eval_batch_step: [0, 1000]
+  cal_metric_during_train: True
+  pretrained_model:
+  checkpoints:
+  save_inference_dir: ./output/SLANet/infer
+  use_visualdl: False
+  infer_img: ppstructure/docs/table/table.jpg
+  # for data or label process
+  character_dict_path: ppocr/utils/dict/table_structure_dict.txt
+  character_type: en
+  max_text_length: &max_text_length 500
+  box_format: &box_format 'xyxy' # 'xywh', 'xyxy', 'xyxyxyxy'
+  infer_mode: False
+  use_sync_bn: True
+  save_res_path: 'output/infer'
+  d2s_train_image_shape: [3, -1, -1]
+  amp_custom_white_list: ['concat', 'elementwise_sub', 'set_value']
+
+Optimizer:
+  name: Adam
+  beta1: 0.9
+  beta2: 0.999
+  clip_norm: 5.0
+  lr:
+    name: Piecewise
+    learning_rate: 0.001
+    decay_epochs : [40, 50]
+    values : [0.001, 0.0001, 0.00005]
+  regularizer:
+    name: 'L2'
+    factor: 0.00000
+
+Architecture:
+  model_type: table
+  algorithm: SLANet
+  Backbone:
+    name: PPLCNet
+    scale: 1.0
+    pretrained: true
+    use_ssld: true
+  Neck:
+    name: CSPPAN
+    out_channels: 96
+  Head:
+    name: SLAHead
+    hidden_size: 256
+    max_text_length: *max_text_length
+    loc_reg_num: &loc_reg_num 4
+
+Loss:
+  name: SLALoss
+  structure_weight: 1.0
+  loc_weight: 2.0
+  loc_loss: smooth_l1
+
+PostProcess:
+  name: TableLabelDecode
+  merge_no_span_structure: &merge_no_span_structure True
+
+Metric:
+  name: TableMetric
+  main_indicator: acc
+  compute_bbox_metric: False
+  loc_reg_num: *loc_reg_num
+  box_format: *box_format
+
+Train:
+  dataset:
+    name: PubTabDataSet
+    data_dir: train_data/table/pubtabnet/train/
+    label_file_list: [train_data/table/pubtabnet/PubTabNet_2.0.0_train.jsonl]
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - TableLabelEncode:
+          learn_empty_box: False
+          merge_no_span_structure: *merge_no_span_structure
+          replace_empty_cell_token: False
+          loc_reg_num: *loc_reg_num
+          max_text_length: *max_text_length
+      - TableBoxEncode:
+          in_box_format: *box_format
+          out_box_format: *box_format
+      - ResizeTableImage:
+          max_len: 488
+      - NormalizeImage:
+          scale: 1./255.
+          mean: [0.485, 0.456, 0.406]
+          std: [0.229, 0.224, 0.225]
+          order: 'hwc'
+      - PaddingTableImage:
+          size: [488, 488]
+      - ToCHWImage:
+      - KeepKeys:
+          keep_keys: [ 'image', 'structure', 'bboxes', 'bbox_masks', 'shape' ]
+  loader:
+    shuffle: True
+    batch_size_per_card: 48
+    drop_last: True
+    num_workers: 1
+
+Eval:
+  dataset:
+    name: PubTabDataSet
+    data_dir: train_data/table/pubtabnet/val/
+    label_file_list: [train_data/table/pubtabnet/PubTabNet_2.0.0_val.jsonl]
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - TableLabelEncode:
+          learn_empty_box: False
+          merge_no_span_structure: *merge_no_span_structure
+          replace_empty_cell_token: False
+          loc_reg_num: *loc_reg_num
+          max_text_length: *max_text_length
+      - TableBoxEncode:
+          in_box_format: *box_format
+          out_box_format: *box_format
+      - ResizeTableImage:
+          max_len: 488
+      - NormalizeImage:
+          scale: 1./255.
+          mean: [0.485, 0.456, 0.406]
+          std: [0.229, 0.224, 0.225]
+          order: 'hwc'
+      - PaddingTableImage:
+          size: [488, 488]
+      - ToCHWImage:
+      - KeepKeys:
+          keep_keys: [ 'image', 'structure', 'bboxes', 'bbox_masks', 'shape' ]
+  loader:
+    shuffle: False
+    drop_last: False
+    batch_size_per_card: 48
+    num_workers: 1
--- a/configs/table/SLANet_ch.yml
+++ b/configs/table/SLANet_ch.yml
+Global:
+  use_gpu: True
+  epoch_num: 400
+  log_smooth_window: 20
+  print_batch_step: 20
+  save_model_dir: ./output/SLANet_ch
+  save_epoch_step: 400
+  # evaluation is run every 331 iterations after the 0th iteration
+  eval_batch_step: [0, 331]
+  cal_metric_during_train: True
+  pretrained_model: 
+  checkpoints: 
+  save_inference_dir: ./output/SLANet_ch/infer
+  use_visualdl: False
+  infer_img: ppstructure/docs/table/table.jpg
+  # for data or label process
+  character_dict_path: ppocr/utils/dict/table_structure_dict_ch.txt
+  character_type: en
+  max_text_length: &max_text_length 500
+  box_format: &box_format xyxyxyxy # 'xywh', 'xyxy', 'xyxyxyxy'
+  infer_mode: False
+  use_sync_bn: True
+  save_res_path: output/infer
+
+Optimizer:
+  name: Adam
+  beta1: 0.9
+  beta2: 0.999
+  clip_norm: 5.0
+  lr:
+    learning_rate: 0.001
+  regularizer:
+    name: 'L2'
+    factor: 0.00000
+
+Architecture:
+  model_type: table
+  algorithm: SLANet
+  Backbone:
+    name: PPLCNet
+    scale: 1.0
+    pretrained: True
+    use_ssld: True
+  Neck:
+    name: CSPPAN
+    out_channels: 96
+  Head:
+    name: SLAHead
+    hidden_size: 256
+    max_text_length: *max_text_length
+    loc_reg_num: &loc_reg_num 8
+
+Loss:
+  name: SLALoss
+  structure_weight: 1.0
+  loc_weight: 2.0
+  loc_loss: smooth_l1
+
+PostProcess:
+  name: TableLabelDecode
+  merge_no_span_structure: &merge_no_span_structure True
+
+Metric:
+  name: TableMetric
+  main_indicator: acc
+  compute_bbox_metric: False
+  loc_reg_num: *loc_reg_num
+  box_format: *box_format
+  del_thead_tbody: True
+
+Train:
+  dataset:
+    name: PubTabDataSet
+    data_dir: train_data/table/train/
+    label_file_list: [train_data/table/train.txt]
+    transforms:
+      - DecodeImage:
+          img_mode: BGR
+          channel_first: False
+      - TableLabelEncode:
+          learn_empty_box: False
+          merge_no_span_structure: *merge_no_span_structure
+          replace_empty_cell_token: False
+          loc_reg_num: *loc_reg_num
+          max_text_length: *max_text_length
+      - TableBoxEncode:
+          in_box_format: *box_format
+          out_box_format: *box_format
+      - ResizeTableImage:
+          max_len: 488
+      - NormalizeImage:
+          scale: 1./255.
+          mean: [0.485, 0.456, 0.406]
+          std: [0.229, 0.224, 0.225]
+          order: 'hwc'
+      - PaddingTableImage:
+          size: [488, 488]
+      - ToCHWImage:
+      - KeepKeys:
+          keep_keys: [ 'image', 'structure', 'bboxes', 'bbox_masks', 'shape' ]
+  loader:
+    shuffle: True
+    batch_size_per_card: 48
+    drop_last: True
+    num_workers: 1
+
+Eval:
+  dataset:
+    name: PubTabDataSet
+    data_dir: train_data/table/val/
+    label_file_list: [train_data/table/val.txt]
+    transforms:
+      - DecodeImage:
+          img_mode: BGR
+          channel_first: False
+      - TableLabelEncode:
+          learn_empty_box: False
+          merge_no_span_structure: *merge_no_span_structure
+          replace_empty_cell_token: False
+          loc_reg_num: *loc_reg_num
+          max_text_length: *max_text_length
+      - TableBoxEncode:
+          in_box_format: *box_format
+          out_box_format: *box_format
+      - ResizeTableImage:
+          max_len: 488
+      - NormalizeImage:
+          scale: 1./255.
+          mean: [0.485, 0.456, 0.406]
+          std: [0.229, 0.224, 0.225]
+          order: 'hwc'
+      - PaddingTableImage:
+          size: [488, 488]
+      - ToCHWImage:
+      - KeepKeys:
+          keep_keys: [ 'image', 'structure', 'bboxes', 'bbox_masks', 'shape' ]
+  loader:
+    shuffle: False
+    drop_last: False
+    batch_size_per_card: 48
+    num_workers: 1
--- a/configs/table/table_master.yml
+++ b/configs/table/table_master.yml
+Global:
+  use_gpu: true
+  epoch_num: 17
+  log_smooth_window: 20
+  print_batch_step: 100
+  save_model_dir: ./output/table_master/
+  save_epoch_step: 17
+  eval_batch_step: [0,  6259]
+  cal_metric_during_train: true
+  pretrained_model: null
+  checkpoints:
+  save_inference_dir: output/table_master/infer
+  use_visualdl: false
+  infer_img: ppstructure/docs/table/table.jpg
+  save_res_path: ./output/table_master
+  character_dict_path: ppocr/utils/dict/table_master_structure_dict.txt
+  infer_mode: false
+  max_text_length: &max_text_length 500
+  box_format: &box_format 'xywh' # 'xywh', 'xyxy', 'xyxyxyxy'
+  d2s_train_image_shape: [3, 480, 480]
+
+
+Optimizer:
+  name: Adam
+  beta1: 0.9
+  beta2: 0.999
+  lr:
+    name: MultiStepDecay
+    learning_rate: 0.001
+    milestones: [12, 15]
+    gamma: 0.1
+    warmup_epoch: 0.02
+  regularizer:
+    name: L2
+    factor: 0.0
+
+Architecture:
+  model_type: table
+  algorithm: TableMaster
+  Backbone:
+    name: TableResNetExtra
+    gcb_config:
+      ratio: 0.0625
+      headers: 1
+      att_scale: False
+      fusion_type: channel_add
+      layers: [False, True, True, True]
+    layers: [1,2,5,3]
+  Head:
+    name: TableMasterHead
+    hidden_size: 512
+    headers: 8
+    dropout: 0
+    d_ff: 2024
+    max_text_length: *max_text_length
+    loc_reg_num: &loc_reg_num 4
+
+Loss:
+  name: TableMasterLoss
+  ignore_index: 42 # set to len of dict + 3
+
+PostProcess:
+  name: TableMasterLabelDecode
+  box_shape: pad
+  merge_no_span_structure: &merge_no_span_structure True
+
+Metric:
+  name: TableMetric
+  main_indicator: acc
+  compute_bbox_metric: False
+  box_format: *box_format
+
+Train:
+  dataset:
+    name: PubTabDataSet
+    data_dir: train_data/table/pubtabnet/train/
+    label_file_list: [train_data/table/pubtabnet/PubTabNet_2.0.0_train.jsonl]
+    transforms:
+      - DecodeImage:
+          img_mode: BGR
+          channel_first: False
+      - TableMasterLabelEncode:
+          learn_empty_box: False
+          merge_no_span_structure: *merge_no_span_structure
+          replace_empty_cell_token: True
+          loc_reg_num: *loc_reg_num
+          max_text_length: *max_text_length
+      - ResizeTableImage:
+          max_len: 480
+          resize_bboxes: True
+      - PaddingTableImage:
+          size: [480, 480]
+      - TableBoxEncode:
+          in_box_format: *box_format
+          out_box_format: *box_format
+      - NormalizeImage:
+          scale: 1./255.
+          mean: [0.5, 0.5, 0.5]
+          std: [0.5, 0.5, 0.5]
+          order: hwc
+      - ToCHWImage: null
+      - KeepKeys:
+          keep_keys: [image, structure, bboxes, bbox_masks, shape]
+  loader:
+    shuffle: True
+    batch_size_per_card: 10
+    drop_last: True
+    num_workers: 8
+
+Eval:
+  dataset:
+    name: PubTabDataSet
+    data_dir: train_data/table/pubtabnet/val/
+    label_file_list: [train_data/table/pubtabnet/PubTabNet_2.0.0_val.jsonl]
+    transforms:
+      - DecodeImage:
+          img_mode: BGR
+          channel_first: False
+      - TableMasterLabelEncode:
+          learn_empty_box: False
+          merge_no_span_structure: *merge_no_span_structure
+          replace_empty_cell_token: True
+          loc_reg_num: *loc_reg_num
+          max_text_length: *max_text_length
+      - ResizeTableImage:
+          max_len: 480
+          resize_bboxes: True
+      - PaddingTableImage:
+          size: [480, 480]
+      - TableBoxEncode:
+          in_box_format: *box_format
+          out_box_format: *box_format
+      - NormalizeImage:
+          scale: 1./255.
+          mean: [0.5, 0.5, 0.5]
+          std: [0.5, 0.5, 0.5]
+          order: hwc
+      - ToCHWImage: null
+      - KeepKeys:
+          keep_keys: [image, structure, bboxes, bbox_masks, shape]
+  loader:
+    shuffle: False
+    drop_last: False
+    batch_size_per_card: 10
+    num_workers: 8
\ No newline at end of file
--- a/configs/table/table_mv3.yml
+++ b/configs/table/table_mv3.yml
+Global:
+  use_gpu: true
+  epoch_num: 400
+  log_smooth_window: 20
+  print_batch_step: 5
+  save_model_dir: ./output/table_mv3/
+  save_epoch_step: 400
+  # evaluation is run every 400 iterations after the 0th iteration
+  eval_batch_step: [0, 400]
+  cal_metric_during_train: True
+  pretrained_model:
+  checkpoints:
+  save_inference_dir:
+  use_visualdl: False
+  infer_img: ppstructure/docs/table/table.jpg
+  save_res_path: output/table_mv3
+  # for data or label process
+  character_dict_path: ppocr/utils/dict/table_structure_dict.txt
+  character_type: en
+  max_text_length: &max_text_length 500
+  box_format: &box_format 'xyxy' # 'xywh', 'xyxy', 'xyxyxyxy'
+  infer_mode: False
+  amp_custom_black_list: ['matmul_v2','elementwise_add']
+
+Optimizer:
+  name: Adam
+  beta1: 0.9
+  beta2: 0.999
+  clip_norm: 5.0
+  lr:
+    learning_rate: 0.001
+  regularizer:
+    name: 'L2'
+    factor: 0.00000
+
+Architecture:
+  model_type: table
+  algorithm: TableAttn
+  Backbone:
+    name: MobileNetV3
+    scale: 1.0
+    model_name: small
+    disable_se: true
+  Head:
+    name: TableAttentionHead
+    hidden_size: 256
+    max_text_length: *max_text_length
+    loc_reg_num: &loc_reg_num 4
+
+Loss:
+  name: TableAttentionLoss
+  structure_weight: 100.0
+  loc_weight: 10000.0
+
+PostProcess:
+  name: TableLabelDecode
+
+Metric:
+  name: TableMetric
+  main_indicator: acc
+  compute_bbox_metric: false # cost many time, set False for training
+
+Train:
+  dataset:
+    name: PubTabDataSet
+    data_dir: train_data/table/pubtabnet/train/
+    label_file_list: [train_data/table/pubtabnet/PubTabNet_2.0.0_train.jsonl]
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - TableLabelEncode:
+          learn_empty_box: False
+          merge_no_span_structure: False
+          replace_empty_cell_token: False
+          loc_reg_num: *loc_reg_num
+          max_text_length: *max_text_length
+      - TableBoxEncode:
+      - ResizeTableImage:
+          max_len: 488
+      - NormalizeImage:
+          scale: 1./255.
+          mean: [0.485, 0.456, 0.406]
+          std: [0.229, 0.224, 0.225]
+          order: 'hwc'
+      - PaddingTableImage:
+          size: [488, 488]
+      - ToCHWImage:
+      - KeepKeys:
+          keep_keys: [ 'image', 'structure', 'bboxes', 'bbox_masks', 'shape' ]
+  loader:
+    shuffle: True
+    batch_size_per_card: 48
+    drop_last: True
+    num_workers: 1
+
+Eval:
+  dataset:
+    name: PubTabDataSet
+    data_dir: train_data/table/pubtabnet/val/
+    label_file_list: [train_data/table/pubtabnet/PubTabNet_2.0.0_val.jsonl]
+    transforms:
+      - DecodeImage: # load image
+          img_mode: BGR
+          channel_first: False
+      - TableLabelEncode:
+          learn_empty_box: False
+          merge_no_span_structure: False
+          replace_empty_cell_token: False
+          loc_reg_num: *loc_reg_num
+          max_text_length: *max_text_length
+      - TableBoxEncode:
+      - ResizeTableImage:
+          max_len: 488
+      - NormalizeImage:
+          scale: 1./255.
+          mean: [0.485, 0.456, 0.406]
+          std: [0.229, 0.224, 0.225]
+          order: 'hwc'
+      - PaddingTableImage:
+          size: [488, 488]
+      - ToCHWImage:
+      - KeepKeys:
+          keep_keys: [ 'image', 'structure', 'bboxes', 'bbox_masks', 'shape' ]
+  loader:
+    shuffle: False
+    drop_last: False
+    batch_size_per_card: 48
+    num_workers: 1
--- a/debug/backup_env.0.json
+++ b/debug/backup_env.0.json
+{
+    "AMDGPU_TARGETS": "gfx906;gfx926",
+    "CPLUS_INCLUDE_PATH": "/opt/dtk/include:/opt/hyhal/include:/opt/dtk/llvm/include:/opt/dtk-23.10/include:/opt/hyhal/include:/opt/dtk-23.10/llvm/include:/opt/dtk/include:/opt/hyhal/include:/opt/dtk/llvm/include:",
+    "CUSTOM_DEVICE_ROOT": "",
+    "C_INCLUDE_PATH": "/opt/dtk/include:/opt/hyhal/include:/opt/dtk/llvm/include:/opt/dtk-23.10/include:/opt/hyhal/include:/opt/dtk-23.10/llvm/include:/opt/dtk/include:/opt/hyhal/include:/opt/dtk/llvm/include:",
+    "DTKROOT": "/opt/dtk",
+    "FLAGS_selected_gpus": "0",
+    "HIP_PATH": "/opt/dtk/hip",
+    "HOME": "/root",
+    "HOSTNAME": "localhost.localdomain",
+    "HYHAL_PATH": "/opt/hyhal",
+    "INFOPATH": "/opt/rh/devtoolset-7/root/usr/share/info",
+    "LANG": "en_US.UTF-8",
+    "LANGUAGE": "en_US.UTF-8",
+    "LC_ALL": "en_US.UTF-8",
+    "LD_LIBRARY_PATH": "/opt/dtk/hip/lib:/opt/dtk/llvm/lib:/opt/dtk/lib:/opt/dtk/lib64:/opt/hyhal/lib:/opt/hyhal/lib64:/opt/dtk-23.10/hip/lib:/opt/dtk-23.10/llvm/lib:/opt/dtk-23.10/lib:/opt/dtk-23.10/lib64:/opt/hyhal/lib:/opt/hyhal/lib64:/usr/local/lib/:/usr/local/lib64/:/opt/mpi/lib:/opt/hwloc/lib:/opt/rh/devtoolset-7/root/usr/lib64:/opt/rh/devtoolset-7/root/usr/lib:/opt/rh/devtoolset-7/root/usr/lib64/dyninst:/opt/rh/devtoolset-7/root/usr/lib/dyninst:/opt/rh/devtoolset-7/root/usr/lib64:/opt/rh/devtoolset-7/root/usr/lib:/usr/local/lib/:/usr/local/lib64/:/opt/mpi/lib:/opt/hwloc/lib:/opt/dtk/hip/lib:/opt/dtk/llvm/lib:/opt/dtk/lib:/opt/dtk/lib64:/opt/hyhal/lib:/opt/hyhal/lib64:/opt/rh/devtoolset-7/root/usr/lib64:/opt/rh/devtoolset-7/root/usr/lib:/opt/rh/devtoolset-7/root/usr/lib64/dyninst:/opt/rh/devtoolset-7/root/usr/lib/dyninst:/opt/rh/devtoolset-7/root/usr/lib64:/opt/rh/devtoolset-7/root/usr/lib:/opt/mpi/lib:/opt/hwloc/lib:/usr/local/lib/:/usr/local/lib64/:",
+    "LESSOPEN": "||/usr/bin/lesspipe.sh %s",
+    "LS_COLORS": "rs=0:di=01;34:ln=01;36:mh=00:pi=40;33:so=01;35:do=01;35:bd=40;33;01:cd=40;33;01:or=40;31;01:mi=01;05;37;41:su=37;41:sg=30;43:ca=30;41:tw=30;42:ow=34;42:st=37;44:ex=01;32:*.tar=01;31:*.tgz=01;31:*.arc=01;31:*.arj=01;31:*.taz=01;31:*.lha=01;31:*.lz4=01;31:*.lzh=01;31:*.lzma=01;31:*.tlz=01;31:*.txz=01;31:*.tzo=01;31:*.t7z=01;31:*.zip=01;31:*.z=01;31:*.Z=01;31:*.dz=01;31:*.gz=01;31:*.lrz=01;31:*.lz=01;31:*.lzo=01;31:*.xz=01;31:*.bz2=01;31:*.bz=01;31:*.tbz=01;31:*.tbz2=01;31:*.tz=01;31:*.deb=01;31:*.rpm=01;31:*.jar=01;31:*.war=01;31:*.ear=01;31:*.sar=01;31:*.rar=01;31:*.alz=01;31:*.ace=01;31:*.zoo=01;31:*.cpio=01;31:*.7z=01;31:*.rz=01;31:*.cab=01;31:*.jpg=01;35:*.jpeg=01;35:*.gif=01;35:*.bmp=01;35:*.pbm=01;35:*.pgm=01;35:*.ppm=01;35:*.tga=01;35:*.xbm=01;35:*.xpm=01;35:*.tif=01;35:*.tiff=01;35:*.png=01;35:*.svg=01;35:*.svgz=01;35:*.mng=01;35:*.pcx=01;35:*.mov=01;35:*.mpg=01;35:*.mpeg=01;35:*.m2v=01;35:*.mkv=01;35:*.webm=01;35:*.ogm=01;35:*.mp4=01;35:*.m4v=01;35:*.mp4v=01;35:*.vob=01;35:*.qt=01;35:*.nuv=01;35:*.wmv=01;35:*.asf=01;35:*.rm=01;35:*.rmvb=01;35:*.flc=01;35:*.avi=01;35:*.fli=01;35:*.flv=01;35:*.gl=01;35:*.dl=01;35:*.xcf=01;35:*.xwd=01;35:*.yuv=01;35:*.cgm=01;35:*.emf=01;35:*.axv=01;35:*.anx=01;35:*.ogv=01;35:*.ogx=01;35:*.aac=01;36:*.au=01;36:*.flac=01;36:*.mid=01;36:*.midi=01;36:*.mka=01;36:*.mp3=01;36:*.mpc=01;36:*.ogg=01;36:*.ra=01;36:*.wav=01;36:*.axa=01;36:*.oga=01;36:*.spx=01;36:*.xspf=01;36:",
+    "MANPATH": "/opt/rh/devtoolset-7/root/usr/share/man:/opt/rh/devtoolset-7/root/usr/share/man:/opt/mpi/share/man:",
+    "MIOPEN_FIND_MODE": "3",
+    "OMP_NUM_THREADS": "1",
+    "PADDLE_CURRENT_ENDPOINT": "127.0.0.1:57585",
+    "PADDLE_GLOBAL_RANK": "0",
+    "PADDLE_GLOBAL_SIZE": "4",
+    "PADDLE_LOCAL_RANK": "0",
+    "PADDLE_LOCAL_SIZE": "4",
+    "PADDLE_LOG_DIR": "/workspace/PaddleOCR-release-2.7/debug",
+    "PADDLE_MASTER": "127.0.0.1:57584",
+    "PADDLE_NNODES": "1",
+    "PADDLE_RANK_IN_NODE": "0",
+    "PADDLE_TRAINERS_NUM": "4",
+    "PADDLE_TRAINER_ENDPOINTS": "127.0.0.1:57585,127.0.0.1:57586,127.0.0.1:57587,127.0.0.1:57588",
+    "PADDLE_TRAINER_ID": "0",
+    "PATH": "/opt/dtk/bin:/opt/dtk/llvm/bin:/opt/dtk/hip/bin:/opt/dtk/hip/bin/hipify:/opt/hyhal/bin:/opt/dtk-23.10/bin:/opt/dtk-23.10/llvm/bin:/opt/dtk-23.10/hip/bin:/opt/dtk-23.10/hip/bin/hipify:/opt/hyhal/bin:/usr/local/bin:/opt/mpi/bin:/opt/hwloc/bin/:/opt/cmake/bin/:/opt/rh/devtoolset-7/root/usr/bin:/usr/local/bin:/opt/mpi/bin:/opt/hwloc/bin/:/opt/cmake/bin/:/opt/dtk/bin:/opt/dtk/llvm/bin:/opt/dtk/hip/bin:/opt/dtk/hip/bin/hipify:/opt/hyhal/bin:/opt/rh/devtoolset-7/root/usr/bin:/opt/mpi/bin:/opt/hwloc/bin/:/opt/cmake/bin/:/usr/local/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin",
+    "PCP_DIR": "/opt/rh/devtoolset-7/root",
+    "PERL5LIB": "/opt/rh/devtoolset-7/root//usr/lib64/perl5/vendor_perl:/opt/rh/devtoolset-7/root/usr/lib/perl5:/opt/rh/devtoolset-7/root//usr/share/perl5/vendor_perl:/opt/rh/devtoolset-7/root//usr/lib64/perl5/vendor_perl:/opt/rh/devtoolset-7/root/usr/lib/perl5:/opt/rh/devtoolset-7/root//usr/share/perl5/vendor_perl",
+    "POD_NAME": "ychmko",
+    "PWD": "/workspace/PaddleOCR-release-2.7",
+    "PYTHONPATH": "/usr/local/lib/python3.8/site-packages:/usr/local/:/opt/rh/devtoolset-7/root/usr/lib64/python2.7/site-packages:/opt/rh/devtoolset-7/root/usr/lib/python2.7/site-packages:/usr/local/lib/python3.8/site-packages:/usr/local/:/opt/rh/devtoolset-7/root/usr/lib64/python2.7/site-packages:/opt/rh/devtoolset-7/root/usr/lib/python2.7/site-packages:/usr/local/lib/python3.8/site-packages:/usr/local/:",
+    "QT_QPA_FONTDIR": "/usr/local/lib/python3.8/site-packages/cv2/qt/fonts",
+    "QT_QPA_PLATFORM_PLUGIN_PATH": "/usr/local/lib/python3.8/site-packages/cv2/qt/plugins",
+    "ROCM_PATH": "/opt/dtk",
+    "SHELL": "/bin/bash",
+    "SHLVL": "2",
+    "TERM": "xterm",
+    "TZ": "Asia/Shanghai",
+    "X_SCLS": "devtoolset-7 ",
+    "_": "/usr/local/bin/python3"
+}
\ No newline at end of file