forked from PaddlePaddle/PaddleOCR
-
Notifications
You must be signed in to change notification settings - Fork 0
/
rec_latex_ocr.yml
128 lines (121 loc) · 2.94 KB
/
rec_latex_ocr.yml
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
Global:
use_gpu: True
epoch_num: 500
log_smooth_window: 20
print_batch_step: 100
save_model_dir: ./output/rec/latex_ocr/
save_epoch_step: 5
max_seq_len: 512
# evaluation is run every 60000 iterations (22 epoch)(batch_size = 56)
eval_batch_step: [0, 60000]
cal_metric_during_train: True
pretrained_model:
checkpoints:
save_inference_dir:
use_visualdl: False
infer_img: doc/datasets/pme_demo/0000013.png
infer_mode: False
use_space_char: False
rec_char_dict_path: ppocr/utils/dict/latex_ocr_tokenizer.json
save_res_path: ./output/rec/predicts_latexocr.txt
Optimizer:
name: AdamW
beta1: 0.9
beta2: 0.999
lr:
name: Const
learning_rate: 0.0001
Architecture:
model_type: rec
algorithm: LaTeXOCR
in_channels: 1
Transform:
Backbone:
name: HybridTransformer
img_size: [192, 672]
patch_size: 16
num_classes: 0
embed_dim: 256
depth: 4
num_heads: 8
input_channel: 1
is_predict: False
is_export: False
Head:
name: LaTeXOCRHead
pad_value: 0
is_export: False
decoder_args:
attn_on_attn: True
cross_attend: True
ff_glu: True
rel_pos_bias: False
use_scalenorm: False
Loss:
name: LaTeXOCRLoss
PostProcess:
name: LaTeXOCRDecode
rec_char_dict_path: ppocr/utils/dict/latex_ocr_tokenizer.json
Metric:
name: LaTeXOCRMetric
main_indicator: exp_rate
cal_blue_score: False
Train:
dataset:
name: LaTeXOCRDataSet
data_dir: ./train_data/LaTeXOCR/train
data: ./train_data/LaTeXOCR/latexocr_train.pkl
min_dimensions: [32, 32]
max_dimensions: [672, 192]
batch_size_per_pair: 56
keep_smaller_batches: False
transforms:
- DecodeImage:
channel_first: False
- MinMaxResize:
min_dimensions: [32, 32]
max_dimensions: [672, 192]
- LatexTrainTransform:
bitmap_prob: .04
- NormalizeImage:
mean: [0.7931, 0.7931, 0.7931]
std: [0.1738, 0.1738, 0.1738]
order: 'hwc'
- LatexImageFormat:
- KeepKeys:
keep_keys: ['image']
loader:
shuffle: True
batch_size_per_card: 1
drop_last: False
num_workers: 0
collate_fn: LaTeXOCRCollator
Eval:
dataset:
name: LaTeXOCRDataSet
data_dir: ./train_data/LaTeXOCR/val
data: ./train_data/LaTeXOCR/latexocr_val.pkl
min_dimensions: [32, 32]
max_dimensions: [672, 192]
batch_size_per_pair: 10
keep_smaller_batches: True
transforms:
- DecodeImage:
channel_first: False
- MinMaxResize:
min_dimensions: [32, 32]
max_dimensions: [672, 192]
- LatexTestTransform:
- NormalizeImage:
mean: [0.7931, 0.7931, 0.7931]
std: [0.1738, 0.1738, 0.1738]
order: 'hwc'
- LatexImageFormat:
- KeepKeys:
keep_keys: ['image']
loader:
shuffle: False
drop_last: False
batch_size_per_card: 1
num_workers: 0
collate_fn: LaTeXOCRCollator