This commit is contained in:
94
ppocr/utils/EN_symbol_dict.txt
Normal file
94
ppocr/utils/EN_symbol_dict.txt
Normal file
@@ -0,0 +1,94 @@
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
!
|
||||
"
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
)
|
||||
*
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
:
|
||||
;
|
||||
<
|
||||
=
|
||||
>
|
||||
?
|
||||
@
|
||||
[
|
||||
\
|
||||
]
|
||||
^
|
||||
_
|
||||
`
|
||||
{
|
||||
|
|
||||
}
|
||||
~
|
||||
13
ppocr/utils/__init__.py
Executable file
13
ppocr/utils/__init__.py
Executable file
@@ -0,0 +1,13 @@
|
||||
# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
5
ppocr/utils/dict/README.md
Normal file
5
ppocr/utils/dict/README.md
Normal file
@@ -0,0 +1,5 @@
|
||||
## Dictionary and Corpus
|
||||
|
||||
Dictionary files (usually character level vocabulary) are included here for easier configuration. Corpus contributed by OSS contributors are listed here, please respect copyrights when using them at your own risk.
|
||||
|
||||
- Burmese corpus: https://github.com/1chimaruGin/BurmeseCorpus
|
||||
117
ppocr/utils/dict/ar_dict.txt
Normal file
117
ppocr/utils/dict/ar_dict.txt
Normal file
@@ -0,0 +1,117 @@
|
||||
a
|
||||
r
|
||||
b
|
||||
i
|
||||
c
|
||||
_
|
||||
m
|
||||
g
|
||||
/
|
||||
1
|
||||
0
|
||||
I
|
||||
L
|
||||
S
|
||||
V
|
||||
R
|
||||
C
|
||||
2
|
||||
v
|
||||
l
|
||||
6
|
||||
3
|
||||
9
|
||||
.
|
||||
j
|
||||
p
|
||||
ا
|
||||
ل
|
||||
م
|
||||
ر
|
||||
ج
|
||||
و
|
||||
ح
|
||||
ي
|
||||
ة
|
||||
5
|
||||
8
|
||||
7
|
||||
أ
|
||||
ب
|
||||
ض
|
||||
4
|
||||
ك
|
||||
س
|
||||
ه
|
||||
ث
|
||||
ن
|
||||
ط
|
||||
ع
|
||||
ت
|
||||
غ
|
||||
خ
|
||||
ف
|
||||
ئ
|
||||
ز
|
||||
إ
|
||||
د
|
||||
ص
|
||||
ظ
|
||||
ذ
|
||||
ش
|
||||
ى
|
||||
ق
|
||||
ؤ
|
||||
آ
|
||||
ء
|
||||
s
|
||||
e
|
||||
n
|
||||
w
|
||||
t
|
||||
u
|
||||
z
|
||||
d
|
||||
A
|
||||
N
|
||||
G
|
||||
h
|
||||
o
|
||||
E
|
||||
T
|
||||
H
|
||||
O
|
||||
B
|
||||
y
|
||||
F
|
||||
U
|
||||
J
|
||||
X
|
||||
W
|
||||
P
|
||||
Z
|
||||
M
|
||||
k
|
||||
q
|
||||
Y
|
||||
Q
|
||||
D
|
||||
f
|
||||
K
|
||||
x
|
||||
'
|
||||
%
|
||||
-
|
||||
#
|
||||
@
|
||||
!
|
||||
&
|
||||
$
|
||||
,
|
||||
:
|
||||
é
|
||||
?
|
||||
+
|
||||
É
|
||||
(
|
||||
|
||||
161
ppocr/utils/dict/arabic_dict.txt
Normal file
161
ppocr/utils/dict/arabic_dict.txt
Normal file
@@ -0,0 +1,161 @@
|
||||
!
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
:
|
||||
?
|
||||
@
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
_
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
É
|
||||
é
|
||||
ء
|
||||
آ
|
||||
أ
|
||||
ؤ
|
||||
إ
|
||||
ئ
|
||||
ا
|
||||
ب
|
||||
ة
|
||||
ت
|
||||
ث
|
||||
ج
|
||||
ح
|
||||
خ
|
||||
د
|
||||
ذ
|
||||
ر
|
||||
ز
|
||||
س
|
||||
ش
|
||||
ص
|
||||
ض
|
||||
ط
|
||||
ظ
|
||||
ع
|
||||
غ
|
||||
ف
|
||||
ق
|
||||
ك
|
||||
ل
|
||||
م
|
||||
ن
|
||||
ه
|
||||
و
|
||||
ى
|
||||
ي
|
||||
ً
|
||||
ٌ
|
||||
ٍ
|
||||
َ
|
||||
ُ
|
||||
ِ
|
||||
ّ
|
||||
ْ
|
||||
ٓ
|
||||
ٔ
|
||||
ٰ
|
||||
ٱ
|
||||
ٹ
|
||||
پ
|
||||
چ
|
||||
ڈ
|
||||
ڑ
|
||||
ژ
|
||||
ک
|
||||
ڭ
|
||||
گ
|
||||
ں
|
||||
ھ
|
||||
ۀ
|
||||
ہ
|
||||
ۂ
|
||||
ۃ
|
||||
ۆ
|
||||
ۇ
|
||||
ۈ
|
||||
ۋ
|
||||
ی
|
||||
ې
|
||||
ے
|
||||
ۓ
|
||||
ە
|
||||
١
|
||||
٢
|
||||
٣
|
||||
٤
|
||||
٥
|
||||
٦
|
||||
٧
|
||||
٨
|
||||
٩
|
||||
145
ppocr/utils/dict/be_dict.txt
Normal file
145
ppocr/utils/dict/be_dict.txt
Normal file
@@ -0,0 +1,145 @@
|
||||
b
|
||||
e
|
||||
_
|
||||
i
|
||||
m
|
||||
g
|
||||
/
|
||||
2
|
||||
0
|
||||
I
|
||||
L
|
||||
S
|
||||
V
|
||||
R
|
||||
C
|
||||
1
|
||||
v
|
||||
a
|
||||
l
|
||||
6
|
||||
9
|
||||
4
|
||||
3
|
||||
.
|
||||
j
|
||||
p
|
||||
п
|
||||
а
|
||||
з
|
||||
б
|
||||
у
|
||||
г
|
||||
н
|
||||
ц
|
||||
ь
|
||||
8
|
||||
м
|
||||
л
|
||||
і
|
||||
о
|
||||
ў
|
||||
ы
|
||||
7
|
||||
5
|
||||
М
|
||||
х
|
||||
с
|
||||
р
|
||||
ф
|
||||
я
|
||||
е
|
||||
д
|
||||
ж
|
||||
ю
|
||||
ч
|
||||
й
|
||||
к
|
||||
Д
|
||||
в
|
||||
Б
|
||||
т
|
||||
І
|
||||
ш
|
||||
ё
|
||||
э
|
||||
К
|
||||
Л
|
||||
Н
|
||||
А
|
||||
Ж
|
||||
Г
|
||||
В
|
||||
П
|
||||
З
|
||||
Е
|
||||
О
|
||||
Р
|
||||
С
|
||||
У
|
||||
Ё
|
||||
Й
|
||||
Т
|
||||
Ч
|
||||
Э
|
||||
Ц
|
||||
Ю
|
||||
Ш
|
||||
Ф
|
||||
Х
|
||||
Я
|
||||
Ь
|
||||
Ы
|
||||
Ў
|
||||
s
|
||||
c
|
||||
n
|
||||
w
|
||||
M
|
||||
o
|
||||
t
|
||||
T
|
||||
E
|
||||
A
|
||||
B
|
||||
u
|
||||
h
|
||||
y
|
||||
k
|
||||
r
|
||||
H
|
||||
d
|
||||
Y
|
||||
O
|
||||
U
|
||||
F
|
||||
f
|
||||
x
|
||||
D
|
||||
G
|
||||
N
|
||||
K
|
||||
P
|
||||
z
|
||||
J
|
||||
X
|
||||
W
|
||||
Z
|
||||
Q
|
||||
%
|
||||
-
|
||||
q
|
||||
@
|
||||
'
|
||||
!
|
||||
#
|
||||
&
|
||||
,
|
||||
:
|
||||
$
|
||||
(
|
||||
?
|
||||
é
|
||||
+
|
||||
É
|
||||
|
||||
74
ppocr/utils/dict/bengali_dict.txt
Normal file
74
ppocr/utils/dict/bengali_dict.txt
Normal file
@@ -0,0 +1,74 @@
|
||||
হ
|
||||
থ
|
||||
শ
|
||||
৫
|
||||
ক
|
||||
ও
|
||||
য
|
||||
০
|
||||
গ
|
||||
দ
|
||||
ড়
|
||||
খ
|
||||
য়
|
||||
ঋ
|
||||
ন
|
||||
অ
|
||||
৪
|
||||
এ
|
||||
ব
|
||||
ঠ
|
||||
ঢ
|
||||
৭
|
||||
৯
|
||||
ধ
|
||||
ঙ
|
||||
ট
|
||||
ঝ
|
||||
ৎ
|
||||
ণ
|
||||
ত
|
||||
র
|
||||
২
|
||||
চ
|
||||
ঌ
|
||||
ড
|
||||
৬
|
||||
ঔ
|
||||
প
|
||||
ভ
|
||||
ম
|
||||
ঢ়
|
||||
ঈ
|
||||
৮
|
||||
ঘ
|
||||
১
|
||||
ষ
|
||||
৩
|
||||
ফ
|
||||
ছ
|
||||
ল
|
||||
জ
|
||||
আ
|
||||
।
|
||||
ঊ
|
||||
ই
|
||||
স
|
||||
ঐ
|
||||
উ
|
||||
ঞ
|
||||
া
|
||||
্
|
||||
ু
|
||||
ী
|
||||
ে
|
||||
ং
|
||||
ি
|
||||
়
|
||||
ঁ
|
||||
ৃ
|
||||
ো
|
||||
ূ
|
||||
ৈ
|
||||
ৌ
|
||||
ঃ
|
||||
140
ppocr/utils/dict/bg_dict.txt
Normal file
140
ppocr/utils/dict/bg_dict.txt
Normal file
@@ -0,0 +1,140 @@
|
||||
!
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
:
|
||||
?
|
||||
@
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
_
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
É
|
||||
é
|
||||
А
|
||||
Б
|
||||
В
|
||||
Г
|
||||
Д
|
||||
Е
|
||||
Ж
|
||||
З
|
||||
И
|
||||
Й
|
||||
К
|
||||
Л
|
||||
М
|
||||
Н
|
||||
О
|
||||
П
|
||||
Р
|
||||
С
|
||||
Т
|
||||
У
|
||||
Ф
|
||||
Х
|
||||
Ц
|
||||
Ч
|
||||
Ш
|
||||
Щ
|
||||
Ъ
|
||||
Ю
|
||||
Я
|
||||
а
|
||||
б
|
||||
в
|
||||
г
|
||||
д
|
||||
е
|
||||
ж
|
||||
з
|
||||
и
|
||||
й
|
||||
к
|
||||
л
|
||||
м
|
||||
н
|
||||
о
|
||||
п
|
||||
р
|
||||
с
|
||||
т
|
||||
у
|
||||
ф
|
||||
х
|
||||
ц
|
||||
ч
|
||||
ш
|
||||
щ
|
||||
ъ
|
||||
ь
|
||||
ю
|
||||
я
|
||||
|
||||
160
ppocr/utils/dict/bm_dict.txt
Normal file
160
ppocr/utils/dict/bm_dict.txt
Normal file
@@ -0,0 +1,160 @@
|
||||
က
|
||||
ခ
|
||||
ဂ
|
||||
ဃ
|
||||
င
|
||||
စ
|
||||
ဆ
|
||||
ဇ
|
||||
ဈ
|
||||
ဉ
|
||||
ည
|
||||
ဋ
|
||||
ဌ
|
||||
ဍ
|
||||
ဎ
|
||||
ဏ
|
||||
တ
|
||||
ထ
|
||||
ဒ
|
||||
ဓ
|
||||
န
|
||||
ပ
|
||||
ဖ
|
||||
ဗ
|
||||
ဘ
|
||||
မ
|
||||
ယ
|
||||
ရ
|
||||
လ
|
||||
ဝ
|
||||
သ
|
||||
ဟ
|
||||
ဠ
|
||||
အ
|
||||
ဢ
|
||||
ဣ
|
||||
ဤ
|
||||
ဥ
|
||||
ဦ
|
||||
ဧ
|
||||
ဨ
|
||||
ဩ
|
||||
ဪ
|
||||
ါ
|
||||
ာ
|
||||
ိ
|
||||
ီ
|
||||
ု
|
||||
ူ
|
||||
ေ
|
||||
ဲ
|
||||
ဳ
|
||||
ဴ
|
||||
ဵ
|
||||
ံ
|
||||
့
|
||||
း
|
||||
္
|
||||
်
|
||||
ျ
|
||||
ြ
|
||||
ွ
|
||||
ှ
|
||||
ဿ
|
||||
၀
|
||||
၁
|
||||
၂
|
||||
၃
|
||||
၄
|
||||
၅
|
||||
၆
|
||||
၇
|
||||
၈
|
||||
၉
|
||||
၊
|
||||
။
|
||||
၌
|
||||
၍
|
||||
၎
|
||||
၏
|
||||
ၐ
|
||||
ၑ
|
||||
ၒ
|
||||
ၓ
|
||||
ၔ
|
||||
ၕ
|
||||
ၖ
|
||||
ၗ
|
||||
ၘ
|
||||
ၙ
|
||||
ၚ
|
||||
ၛ
|
||||
ၜ
|
||||
ၝ
|
||||
ၞ
|
||||
ၟ
|
||||
ၠ
|
||||
ၡ
|
||||
ၢ
|
||||
ၣ
|
||||
ၤ
|
||||
ၥ
|
||||
ၦ
|
||||
ၧ
|
||||
ၨ
|
||||
ၩ
|
||||
ၪ
|
||||
ၫ
|
||||
ၬ
|
||||
ၭ
|
||||
ၮ
|
||||
ၯ
|
||||
ၰ
|
||||
ၱ
|
||||
ၲ
|
||||
ၳ
|
||||
ၴ
|
||||
ၵ
|
||||
ၶ
|
||||
ၷ
|
||||
ၸ
|
||||
ၹ
|
||||
ၺ
|
||||
ၻ
|
||||
ၼ
|
||||
ၽ
|
||||
ၾ
|
||||
ၿ
|
||||
ႀ
|
||||
ႁ
|
||||
ႂ
|
||||
ႃ
|
||||
ႄ
|
||||
ႅ
|
||||
ႆ
|
||||
ႇ
|
||||
ႈ
|
||||
ႉ
|
||||
ႊ
|
||||
ႋ
|
||||
ႌ
|
||||
ႍ
|
||||
ႎ
|
||||
ႏ
|
||||
႐
|
||||
႑
|
||||
႒
|
||||
႓
|
||||
႔
|
||||
႕
|
||||
႖
|
||||
႗
|
||||
႘
|
||||
႙
|
||||
ႚ
|
||||
ႛ
|
||||
ႜ
|
||||
ႝ
|
||||
႞
|
||||
႟
|
||||
3219
ppocr/utils/dict/bm_dict_add.txt
Normal file
3219
ppocr/utils/dict/bm_dict_add.txt
Normal file
File diff suppressed because it is too large
Load Diff
477
ppocr/utils/dict/bn_dict.txt
Normal file
477
ppocr/utils/dict/bn_dict.txt
Normal file
@@ -0,0 +1,477 @@
|
||||
|
||||
!
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
+
|
||||
|
||||
-
|
||||
.
|
||||
/
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
:
|
||||
?
|
||||
@
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
_
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
É
|
||||
é
|
||||
'
|
||||
"
|
||||
ঃ
|
||||
,
|
||||
;
|
||||
!
|
||||
?
|
||||
।
|
||||
—
|
||||
ঃ-
|
||||
-
|
||||
(
|
||||
)
|
||||
[
|
||||
]
|
||||
{
|
||||
}
|
||||
√
|
||||
...
|
||||
.
|
||||
/
|
||||
\
|
||||
=
|
||||
<
|
||||
>
|
||||
৳
|
||||
°
|
||||
%
|
||||
অ
|
||||
আ
|
||||
ই
|
||||
ঈ
|
||||
উ
|
||||
ঊ
|
||||
ঋ
|
||||
এ
|
||||
ঐ
|
||||
ও
|
||||
ঔ
|
||||
্
|
||||
া
|
||||
ি
|
||||
ী
|
||||
ু
|
||||
ূ
|
||||
ে
|
||||
ৈ
|
||||
ো
|
||||
ৌ
|
||||
ৃ
|
||||
্য
|
||||
্র
|
||||
্ব
|
||||
ক
|
||||
খ
|
||||
গ
|
||||
ঘ
|
||||
ঙ
|
||||
চ
|
||||
ছ
|
||||
জ
|
||||
ঝ
|
||||
ঞ
|
||||
ট
|
||||
ঠ
|
||||
ড
|
||||
ঢ
|
||||
ণ
|
||||
ত
|
||||
থ
|
||||
দ
|
||||
ধ
|
||||
ন
|
||||
প
|
||||
ফ
|
||||
ব
|
||||
ভ
|
||||
ম
|
||||
য
|
||||
র
|
||||
ল
|
||||
শ
|
||||
ষ
|
||||
স
|
||||
হ
|
||||
ড়
|
||||
ঢ়
|
||||
য়
|
||||
ং
|
||||
ৎ
|
||||
ঁ
|
||||
ক্ক
|
||||
ক্ট
|
||||
ক্ত
|
||||
ত্র
|
||||
ক্ব
|
||||
ক্ম
|
||||
ক্য
|
||||
ক্র
|
||||
ক্ল
|
||||
ক্ষ
|
||||
ক্ষ্ণ
|
||||
ক্ষ্ব
|
||||
ক্ষ্ম
|
||||
ক্ষ্য
|
||||
ক্স
|
||||
র্ক
|
||||
র্ক্য
|
||||
খ্য
|
||||
খ্র
|
||||
র্খ
|
||||
গ্ণ
|
||||
গ্ধ
|
||||
গ্ধ্য
|
||||
গ্ধ্র
|
||||
গ্ন
|
||||
গ্ন্য
|
||||
গ্ব
|
||||
গ্ম
|
||||
গ্য
|
||||
গ্র
|
||||
গ্র্য
|
||||
গ্ল
|
||||
র্গ
|
||||
র্গ্য
|
||||
র্গ্র
|
||||
ঘ্ন
|
||||
ঘ্য
|
||||
ঘ্র
|
||||
র্ঘ্য
|
||||
র্ঘ
|
||||
ঙ্ক
|
||||
ঙ্ক্য
|
||||
ঙ্ক্ষ
|
||||
ঙ্খ
|
||||
ঙ্খ্য
|
||||
ঙ্গ
|
||||
ঙ্গ্য
|
||||
ঙ্ঘ
|
||||
ঙ্ঘ্য
|
||||
ঙ্ঘ্র
|
||||
ঙ্ম
|
||||
র্ঙ্গ
|
||||
চ্চ
|
||||
চ্ছ
|
||||
চ্ছ্ব
|
||||
চ্ছ্র
|
||||
চ্ঞ
|
||||
চ্ব
|
||||
চ্য
|
||||
র্চ্য
|
||||
র্চ
|
||||
র্ছ
|
||||
জ্জ
|
||||
জ্জ্ব
|
||||
জ্ঝ
|
||||
জ্ঞ
|
||||
জ্ব
|
||||
জ্য
|
||||
জ্র
|
||||
র্জ্য
|
||||
র্জ্জ
|
||||
র্জ্ঞ
|
||||
র্জ
|
||||
র্ঝ
|
||||
ঞ্চ
|
||||
ঞ্ছ
|
||||
ঞ্জ
|
||||
ঞ্ঝ
|
||||
ট্ট
|
||||
ট্ব
|
||||
ট্ম
|
||||
ট্য
|
||||
ট্র
|
||||
র্ট
|
||||
ড্ড
|
||||
ড্ব
|
||||
ড্ম
|
||||
ড্য
|
||||
ড্র
|
||||
র্ড
|
||||
ঢ্য
|
||||
ঢ্র
|
||||
র্ঢ্য
|
||||
ণ্ট
|
||||
ণ্ঠ
|
||||
ণ্ঠ্য
|
||||
ণ্ড
|
||||
ণ্ড্য
|
||||
ণ্ড্র
|
||||
ণ্ঢ
|
||||
ণ্ণ
|
||||
ণ্ব
|
||||
ণ্ম
|
||||
ণ্য
|
||||
র্ণ্য
|
||||
র্ণ
|
||||
ত্ত
|
||||
ত্ত্ব
|
||||
ত্ত্য
|
||||
ত্থ
|
||||
ত্ন
|
||||
ত্ব
|
||||
ত্ম
|
||||
ত্ম্য
|
||||
ত্য
|
||||
ত্র
|
||||
ত্র্য
|
||||
র্ত্য
|
||||
র্ত
|
||||
র্ত্ম
|
||||
র্ত্র
|
||||
থ্ব
|
||||
থ্য
|
||||
থ্র
|
||||
র্থ্য
|
||||
র্থ
|
||||
দ্গ
|
||||
দ্ঘ
|
||||
দ্দ
|
||||
দ্দ্ব
|
||||
দ্ধ
|
||||
দ্ব
|
||||
দ্ভ
|
||||
দ্ভ্র
|
||||
দ্ম
|
||||
দ্য
|
||||
দ্র
|
||||
দ্র্য
|
||||
র্দ
|
||||
র্দ্ব
|
||||
র্দ্র
|
||||
ধ্ন
|
||||
ধ্ব
|
||||
ধ্ম
|
||||
ধ্য
|
||||
ধ্র
|
||||
র্ধ
|
||||
র্ধ্ব
|
||||
ন্ট
|
||||
ন্ট্র
|
||||
ন্ঠ
|
||||
ন্ড
|
||||
ন্ড্র
|
||||
ন্ত
|
||||
ন্ত্ব
|
||||
ন্ত্য
|
||||
ন্ত্র
|
||||
ন্ত্র্য
|
||||
ন্থ
|
||||
ন্থ্র
|
||||
ন্দ
|
||||
ন্দ্য
|
||||
ন্দ্ব
|
||||
ন্দ্র
|
||||
ন্ধ
|
||||
ন্ধ্য
|
||||
ন্ধ্র
|
||||
ন্ন
|
||||
ন্ব
|
||||
ন্ম
|
||||
ন্য
|
||||
র্ন
|
||||
প্ট
|
||||
প্ত
|
||||
প্ন
|
||||
প্প
|
||||
প্য
|
||||
প্র
|
||||
প্র্য
|
||||
প্ল
|
||||
প্স
|
||||
র্প
|
||||
ফ্র
|
||||
ফ্ল
|
||||
র্ফ
|
||||
ব্জ
|
||||
ব্দ
|
||||
ব্ধ
|
||||
ব্ব
|
||||
ব্য
|
||||
ব্র
|
||||
ব্ল
|
||||
র্ব্য
|
||||
র্ব
|
||||
ভ্ব
|
||||
ভ্য
|
||||
ভ্র
|
||||
ভ্ল
|
||||
র্ভ
|
||||
ম্ন
|
||||
ম্প
|
||||
ম্প্র
|
||||
ম্ফ
|
||||
ম্ব
|
||||
ম্ব্র
|
||||
ম্ভ
|
||||
ম্ভ্র
|
||||
ম্ম
|
||||
ম্য
|
||||
ম্র
|
||||
ম্ল
|
||||
র্ম্য
|
||||
র্ম
|
||||
য্য
|
||||
র্য
|
||||
ল্ক
|
||||
ল্ক্য
|
||||
ল্গ
|
||||
ল্ট
|
||||
ল্ড
|
||||
ল্প
|
||||
ল্ফ
|
||||
ল্ব
|
||||
ল্ভ
|
||||
ল্ম
|
||||
ল্য
|
||||
ল্ল
|
||||
র্ল
|
||||
শ্চ
|
||||
শ্ছ
|
||||
শ্ন
|
||||
শ্ব
|
||||
শ্ম
|
||||
শ্য
|
||||
শ্র
|
||||
শ্ল
|
||||
র্শ্য
|
||||
র্শ
|
||||
র্শ্ব
|
||||
ষ্ক
|
||||
ষ্ক্ব
|
||||
ষ্ক্র
|
||||
ষ্ট
|
||||
ষ্ট্য
|
||||
ষ্ট্র
|
||||
ষ্ঠ
|
||||
ষ্ঠ্য
|
||||
ষ্ণ
|
||||
ষ্ণ্ব
|
||||
ষ্প
|
||||
ষ্প্র
|
||||
ষ্ফ
|
||||
ষ্ব
|
||||
ষ্ম
|
||||
ষ্য
|
||||
র্ষ্য
|
||||
র্ষ
|
||||
র্ষ্ট
|
||||
র্ষ্ণ
|
||||
র্ষ্ণ্য
|
||||
স্ক
|
||||
স্ক্র
|
||||
স্খ
|
||||
স্ট
|
||||
স্ট্র
|
||||
স্ত
|
||||
স্ত্ব
|
||||
স্ত্য
|
||||
স্ত্র
|
||||
স্থ
|
||||
স্থ্য
|
||||
স্ন
|
||||
স্ন্য
|
||||
স্প
|
||||
স্প্র
|
||||
স্প্ল
|
||||
স্ফ
|
||||
স্ব
|
||||
স্ম
|
||||
স্য
|
||||
স্র
|
||||
স্ল
|
||||
স্ক্ল
|
||||
র্স
|
||||
হ্ণ
|
||||
হ্ন
|
||||
হ্ব
|
||||
হ্ম
|
||||
হ্য
|
||||
হ্র
|
||||
হ্ল
|
||||
র্হ্য
|
||||
র্হ
|
||||
ড়্গ
|
||||
র্ৎ
|
||||
০
|
||||
১
|
||||
২
|
||||
৩
|
||||
৪
|
||||
৫
|
||||
৬
|
||||
৭
|
||||
৮
|
||||
৯
|
||||
8421
ppocr/utils/dict/chinese_cht_dict.txt
Normal file
8421
ppocr/utils/dict/chinese_cht_dict.txt
Normal file
File diff suppressed because it is too large
Load Diff
BIN
ppocr/utils/dict/confuse.pkl
Normal file
BIN
ppocr/utils/dict/confuse.pkl
Normal file
Binary file not shown.
163
ppocr/utils/dict/cyrillic_dict.txt
Normal file
163
ppocr/utils/dict/cyrillic_dict.txt
Normal file
@@ -0,0 +1,163 @@
|
||||
|
||||
!
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
:
|
||||
?
|
||||
@
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
_
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
É
|
||||
é
|
||||
Ё
|
||||
Є
|
||||
І
|
||||
Ј
|
||||
Љ
|
||||
Ў
|
||||
А
|
||||
Б
|
||||
В
|
||||
Г
|
||||
Д
|
||||
Е
|
||||
Ж
|
||||
З
|
||||
И
|
||||
Й
|
||||
К
|
||||
Л
|
||||
М
|
||||
Н
|
||||
О
|
||||
П
|
||||
Р
|
||||
С
|
||||
Т
|
||||
У
|
||||
Ф
|
||||
Х
|
||||
Ц
|
||||
Ч
|
||||
Ш
|
||||
Щ
|
||||
Ъ
|
||||
Ы
|
||||
Ь
|
||||
Э
|
||||
Ю
|
||||
Я
|
||||
а
|
||||
б
|
||||
в
|
||||
г
|
||||
д
|
||||
е
|
||||
ж
|
||||
з
|
||||
и
|
||||
й
|
||||
к
|
||||
л
|
||||
м
|
||||
н
|
||||
о
|
||||
п
|
||||
р
|
||||
с
|
||||
т
|
||||
у
|
||||
ф
|
||||
х
|
||||
ц
|
||||
ч
|
||||
ш
|
||||
щ
|
||||
ъ
|
||||
ы
|
||||
ь
|
||||
э
|
||||
ю
|
||||
я
|
||||
ё
|
||||
ђ
|
||||
є
|
||||
і
|
||||
ј
|
||||
љ
|
||||
њ
|
||||
ћ
|
||||
ў
|
||||
џ
|
||||
Ґ
|
||||
ґ
|
||||
167
ppocr/utils/dict/devanagari_dict.txt
Normal file
167
ppocr/utils/dict/devanagari_dict.txt
Normal file
@@ -0,0 +1,167 @@
|
||||
|
||||
!
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
:
|
||||
?
|
||||
@
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
_
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
É
|
||||
é
|
||||
ँ
|
||||
ं
|
||||
ः
|
||||
अ
|
||||
आ
|
||||
इ
|
||||
ई
|
||||
उ
|
||||
ऊ
|
||||
ऋ
|
||||
ए
|
||||
ऐ
|
||||
ऑ
|
||||
ओ
|
||||
औ
|
||||
क
|
||||
ख
|
||||
ग
|
||||
घ
|
||||
ङ
|
||||
च
|
||||
छ
|
||||
ज
|
||||
झ
|
||||
ञ
|
||||
ट
|
||||
ठ
|
||||
ड
|
||||
ढ
|
||||
ण
|
||||
त
|
||||
थ
|
||||
द
|
||||
ध
|
||||
न
|
||||
ऩ
|
||||
प
|
||||
फ
|
||||
ब
|
||||
भ
|
||||
म
|
||||
य
|
||||
र
|
||||
ऱ
|
||||
ल
|
||||
ळ
|
||||
व
|
||||
श
|
||||
ष
|
||||
स
|
||||
ह
|
||||
़
|
||||
ा
|
||||
ि
|
||||
ी
|
||||
ु
|
||||
ू
|
||||
ृ
|
||||
ॅ
|
||||
े
|
||||
ै
|
||||
ॉ
|
||||
ो
|
||||
ौ
|
||||
्
|
||||
॒
|
||||
क़
|
||||
ख़
|
||||
ग़
|
||||
ज़
|
||||
ड़
|
||||
ढ़
|
||||
फ़
|
||||
ॠ
|
||||
।
|
||||
०
|
||||
१
|
||||
२
|
||||
३
|
||||
४
|
||||
५
|
||||
६
|
||||
७
|
||||
८
|
||||
९
|
||||
॰
|
||||
63
ppocr/utils/dict/en_dict.txt
Normal file
63
ppocr/utils/dict/en_dict.txt
Normal file
@@ -0,0 +1,63 @@
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
|
||||
136
ppocr/utils/dict/fa_dict.txt
Normal file
136
ppocr/utils/dict/fa_dict.txt
Normal file
@@ -0,0 +1,136 @@
|
||||
f
|
||||
a
|
||||
_
|
||||
i
|
||||
m
|
||||
g
|
||||
/
|
||||
1
|
||||
3
|
||||
I
|
||||
L
|
||||
S
|
||||
V
|
||||
R
|
||||
C
|
||||
2
|
||||
0
|
||||
v
|
||||
l
|
||||
6
|
||||
8
|
||||
5
|
||||
.
|
||||
j
|
||||
p
|
||||
و
|
||||
د
|
||||
ر
|
||||
ك
|
||||
ن
|
||||
ش
|
||||
ه
|
||||
ا
|
||||
4
|
||||
9
|
||||
ی
|
||||
ج
|
||||
ِ
|
||||
7
|
||||
غ
|
||||
ل
|
||||
س
|
||||
ز
|
||||
ّ
|
||||
ت
|
||||
ک
|
||||
گ
|
||||
ي
|
||||
م
|
||||
ب
|
||||
ف
|
||||
چ
|
||||
خ
|
||||
ق
|
||||
ژ
|
||||
آ
|
||||
ص
|
||||
پ
|
||||
َ
|
||||
ع
|
||||
ئ
|
||||
ح
|
||||
ٔ
|
||||
ض
|
||||
ُ
|
||||
ذ
|
||||
أ
|
||||
ى
|
||||
ط
|
||||
ظ
|
||||
ث
|
||||
ة
|
||||
ً
|
||||
ء
|
||||
ؤ
|
||||
ْ
|
||||
ۀ
|
||||
إ
|
||||
ٍ
|
||||
ٌ
|
||||
ٰ
|
||||
ٓ
|
||||
ٱ
|
||||
s
|
||||
c
|
||||
e
|
||||
n
|
||||
w
|
||||
N
|
||||
E
|
||||
W
|
||||
Y
|
||||
D
|
||||
O
|
||||
H
|
||||
A
|
||||
d
|
||||
z
|
||||
r
|
||||
T
|
||||
G
|
||||
o
|
||||
t
|
||||
x
|
||||
h
|
||||
b
|
||||
B
|
||||
M
|
||||
Z
|
||||
u
|
||||
P
|
||||
F
|
||||
y
|
||||
q
|
||||
U
|
||||
K
|
||||
k
|
||||
J
|
||||
Q
|
||||
'
|
||||
X
|
||||
#
|
||||
?
|
||||
%
|
||||
$
|
||||
,
|
||||
:
|
||||
&
|
||||
!
|
||||
-
|
||||
(
|
||||
É
|
||||
@
|
||||
é
|
||||
+
|
||||
|
||||
136
ppocr/utils/dict/french_dict.txt
Normal file
136
ppocr/utils/dict/french_dict.txt
Normal file
@@ -0,0 +1,136 @@
|
||||
f
|
||||
e
|
||||
n
|
||||
c
|
||||
h
|
||||
_
|
||||
i
|
||||
m
|
||||
g
|
||||
/
|
||||
r
|
||||
v
|
||||
a
|
||||
l
|
||||
t
|
||||
w
|
||||
o
|
||||
d
|
||||
6
|
||||
1
|
||||
.
|
||||
p
|
||||
B
|
||||
u
|
||||
2
|
||||
à
|
||||
3
|
||||
R
|
||||
y
|
||||
4
|
||||
U
|
||||
E
|
||||
A
|
||||
5
|
||||
P
|
||||
O
|
||||
S
|
||||
T
|
||||
D
|
||||
7
|
||||
Z
|
||||
8
|
||||
I
|
||||
N
|
||||
L
|
||||
G
|
||||
M
|
||||
H
|
||||
0
|
||||
J
|
||||
K
|
||||
-
|
||||
9
|
||||
F
|
||||
C
|
||||
V
|
||||
é
|
||||
X
|
||||
'
|
||||
s
|
||||
Q
|
||||
:
|
||||
è
|
||||
x
|
||||
b
|
||||
Y
|
||||
Œ
|
||||
É
|
||||
z
|
||||
W
|
||||
Ç
|
||||
È
|
||||
k
|
||||
Ô
|
||||
ô
|
||||
€
|
||||
À
|
||||
Ê
|
||||
q
|
||||
ù
|
||||
°
|
||||
ê
|
||||
î
|
||||
*
|
||||
Â
|
||||
j
|
||||
"
|
||||
,
|
||||
â
|
||||
%
|
||||
û
|
||||
ç
|
||||
ü
|
||||
?
|
||||
!
|
||||
;
|
||||
ö
|
||||
(
|
||||
)
|
||||
ï
|
||||
º
|
||||
ó
|
||||
ø
|
||||
å
|
||||
+
|
||||
™
|
||||
á
|
||||
Ë
|
||||
<
|
||||
²
|
||||
Á
|
||||
Î
|
||||
&
|
||||
@
|
||||
œ
|
||||
ε
|
||||
Ü
|
||||
ë
|
||||
[
|
||||
]
|
||||
í
|
||||
ò
|
||||
Ö
|
||||
ä
|
||||
ß
|
||||
«
|
||||
»
|
||||
ú
|
||||
ñ
|
||||
æ
|
||||
µ
|
||||
³
|
||||
Å
|
||||
$
|
||||
#
|
||||
|
||||
143
ppocr/utils/dict/german_dict.txt
Normal file
143
ppocr/utils/dict/german_dict.txt
Normal file
@@ -0,0 +1,143 @@
|
||||
|
||||
!
|
||||
"
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
)
|
||||
*
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
:
|
||||
;
|
||||
=
|
||||
>
|
||||
?
|
||||
@
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
[
|
||||
]
|
||||
_
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
£
|
||||
§
|
||||
|
||||
°
|
||||
´
|
||||
µ
|
||||
·
|
||||
º
|
||||
¿
|
||||
Á
|
||||
Ä
|
||||
Å
|
||||
É
|
||||
Ï
|
||||
Ô
|
||||
Ö
|
||||
Ü
|
||||
ß
|
||||
à
|
||||
á
|
||||
â
|
||||
ã
|
||||
ä
|
||||
å
|
||||
æ
|
||||
ç
|
||||
è
|
||||
é
|
||||
ê
|
||||
ë
|
||||
í
|
||||
ï
|
||||
ñ
|
||||
ò
|
||||
ó
|
||||
ô
|
||||
ö
|
||||
ø
|
||||
ù
|
||||
ú
|
||||
û
|
||||
ü
|
||||
ō
|
||||
Š
|
||||
Ÿ
|
||||
ʒ
|
||||
β
|
||||
δ
|
||||
з
|
||||
Ṡ
|
||||
‘
|
||||
€
|
||||
©
|
||||
ª
|
||||
«
|
||||
¬
|
||||
48
ppocr/utils/dict/gujarati_dict.txt
Normal file
48
ppocr/utils/dict/gujarati_dict.txt
Normal file
@@ -0,0 +1,48 @@
|
||||
અ
|
||||
આ
|
||||
ઇ
|
||||
ઈ
|
||||
ઉ
|
||||
ઊ
|
||||
ઋ
|
||||
ઌ
|
||||
એ
|
||||
ઐ
|
||||
ઓ
|
||||
ઔ
|
||||
અં
|
||||
અઃ
|
||||
ક
|
||||
ખ
|
||||
ગ
|
||||
ઘ
|
||||
ઙ
|
||||
ચ
|
||||
છ
|
||||
જ
|
||||
ઝ
|
||||
ઞ
|
||||
ટ
|
||||
ઠ
|
||||
ડ
|
||||
ઢ
|
||||
ણ
|
||||
ત
|
||||
થ
|
||||
દ
|
||||
ધ
|
||||
ન
|
||||
પ
|
||||
ફ
|
||||
બ
|
||||
ભ
|
||||
મ
|
||||
ય
|
||||
ર
|
||||
લ
|
||||
ળ
|
||||
વ
|
||||
શ
|
||||
ષ
|
||||
સ
|
||||
હ
|
||||
214
ppocr/utils/dict/hebrew_dict.txt
Normal file
214
ppocr/utils/dict/hebrew_dict.txt
Normal file
@@ -0,0 +1,214 @@
|
||||
!
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
:
|
||||
?
|
||||
@
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
_
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
É
|
||||
é
|
||||
֑
|
||||
֒
|
||||
֓
|
||||
֔
|
||||
֕
|
||||
֖
|
||||
֗
|
||||
֘
|
||||
֙
|
||||
֚
|
||||
֛
|
||||
֜
|
||||
֝
|
||||
֞
|
||||
֟
|
||||
֠
|
||||
֡
|
||||
֢
|
||||
֣
|
||||
֤
|
||||
֥
|
||||
֦
|
||||
֧
|
||||
֨
|
||||
֩
|
||||
֪
|
||||
֫
|
||||
֬
|
||||
֭
|
||||
֮
|
||||
֯
|
||||
ְ
|
||||
ֱ
|
||||
ֲ
|
||||
ֳ
|
||||
ִ
|
||||
ֵ
|
||||
ֶ
|
||||
ַ
|
||||
ָ
|
||||
ֹ
|
||||
ֺ
|
||||
ֻ
|
||||
ּ
|
||||
ֽ
|
||||
־
|
||||
ֿ
|
||||
׀
|
||||
ׁ
|
||||
ׂ
|
||||
׃
|
||||
ׄ
|
||||
ׅ
|
||||
׆
|
||||
ׇ
|
||||
א
|
||||
ב
|
||||
ג
|
||||
ד
|
||||
ה
|
||||
ו
|
||||
ז
|
||||
ח
|
||||
ט
|
||||
י
|
||||
ך
|
||||
כ
|
||||
ל
|
||||
ם
|
||||
מ
|
||||
ן
|
||||
נ
|
||||
ס
|
||||
ע
|
||||
ף
|
||||
פ
|
||||
ץ
|
||||
צ
|
||||
ק
|
||||
ר
|
||||
ש
|
||||
ת
|
||||
ׯ
|
||||
װ
|
||||
ױ
|
||||
ײ
|
||||
׳
|
||||
״
|
||||
יִ
|
||||
ﬞ
|
||||
ײַ
|
||||
ﬠ
|
||||
ﬡ
|
||||
ﬢ
|
||||
ﬣ
|
||||
ﬤ
|
||||
ﬥ
|
||||
ﬦ
|
||||
ﬧ
|
||||
ﬨ
|
||||
﬩
|
||||
שׁ
|
||||
שׂ
|
||||
שּׁ
|
||||
שּׂ
|
||||
אַ
|
||||
אָ
|
||||
אּ
|
||||
בּ
|
||||
גּ
|
||||
דּ
|
||||
הּ
|
||||
וּ
|
||||
זּ
|
||||
טּ
|
||||
יּ
|
||||
ךּ
|
||||
כּ
|
||||
לּ
|
||||
מּ
|
||||
נּ
|
||||
סּ
|
||||
ףּ
|
||||
פּ
|
||||
צּ
|
||||
קּ
|
||||
רּ
|
||||
שּ
|
||||
תּ
|
||||
וֹ
|
||||
בֿ
|
||||
כֿ
|
||||
פֿ
|
||||
ﭏ
|
||||
162
ppocr/utils/dict/hi_dict.txt
Normal file
162
ppocr/utils/dict/hi_dict.txt
Normal file
@@ -0,0 +1,162 @@
|
||||
|
||||
!
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
:
|
||||
?
|
||||
@
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
_
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
É
|
||||
é
|
||||
ँ
|
||||
ं
|
||||
ः
|
||||
अ
|
||||
आ
|
||||
इ
|
||||
ई
|
||||
उ
|
||||
ऊ
|
||||
ऋ
|
||||
ए
|
||||
ऐ
|
||||
ऑ
|
||||
ओ
|
||||
औ
|
||||
क
|
||||
ख
|
||||
ग
|
||||
घ
|
||||
ङ
|
||||
च
|
||||
छ
|
||||
ज
|
||||
झ
|
||||
ञ
|
||||
ट
|
||||
ठ
|
||||
ड
|
||||
ढ
|
||||
ण
|
||||
त
|
||||
थ
|
||||
द
|
||||
ध
|
||||
न
|
||||
प
|
||||
फ
|
||||
ब
|
||||
भ
|
||||
म
|
||||
य
|
||||
र
|
||||
ल
|
||||
ळ
|
||||
व
|
||||
श
|
||||
ष
|
||||
स
|
||||
ह
|
||||
़
|
||||
ा
|
||||
ि
|
||||
ी
|
||||
ु
|
||||
ू
|
||||
ृ
|
||||
ॅ
|
||||
े
|
||||
ै
|
||||
ॉ
|
||||
ो
|
||||
ौ
|
||||
्
|
||||
क़
|
||||
ख़
|
||||
ग़
|
||||
ज़
|
||||
ड़
|
||||
ढ़
|
||||
फ़
|
||||
०
|
||||
१
|
||||
२
|
||||
३
|
||||
४
|
||||
५
|
||||
६
|
||||
७
|
||||
८
|
||||
९
|
||||
॰
|
||||
118
ppocr/utils/dict/it_dict.txt
Normal file
118
ppocr/utils/dict/it_dict.txt
Normal file
@@ -0,0 +1,118 @@
|
||||
i
|
||||
t
|
||||
_
|
||||
m
|
||||
g
|
||||
/
|
||||
5
|
||||
I
|
||||
L
|
||||
S
|
||||
V
|
||||
R
|
||||
C
|
||||
2
|
||||
0
|
||||
1
|
||||
v
|
||||
a
|
||||
l
|
||||
7
|
||||
8
|
||||
9
|
||||
6
|
||||
.
|
||||
j
|
||||
p
|
||||
|
||||
e
|
||||
r
|
||||
o
|
||||
d
|
||||
s
|
||||
n
|
||||
3
|
||||
4
|
||||
P
|
||||
u
|
||||
c
|
||||
A
|
||||
-
|
||||
,
|
||||
"
|
||||
z
|
||||
h
|
||||
f
|
||||
b
|
||||
q
|
||||
ì
|
||||
'
|
||||
à
|
||||
O
|
||||
è
|
||||
G
|
||||
ù
|
||||
é
|
||||
ò
|
||||
;
|
||||
F
|
||||
E
|
||||
B
|
||||
N
|
||||
H
|
||||
k
|
||||
:
|
||||
U
|
||||
T
|
||||
X
|
||||
D
|
||||
K
|
||||
?
|
||||
[
|
||||
M
|
||||
|
||||
x
|
||||
y
|
||||
(
|
||||
)
|
||||
W
|
||||
ö
|
||||
º
|
||||
w
|
||||
]
|
||||
Q
|
||||
J
|
||||
+
|
||||
ü
|
||||
!
|
||||
È
|
||||
á
|
||||
%
|
||||
=
|
||||
»
|
||||
ñ
|
||||
Ö
|
||||
Y
|
||||
ä
|
||||
í
|
||||
Z
|
||||
«
|
||||
@
|
||||
ó
|
||||
ø
|
||||
ï
|
||||
ú
|
||||
ê
|
||||
ç
|
||||
Á
|
||||
É
|
||||
Å
|
||||
ß
|
||||
{
|
||||
}
|
||||
&
|
||||
`
|
||||
û
|
||||
î
|
||||
#
|
||||
$
|
||||
4399
ppocr/utils/dict/japan_dict.txt
Normal file
4399
ppocr/utils/dict/japan_dict.txt
Normal file
File diff suppressed because it is too large
Load Diff
153
ppocr/utils/dict/ka_dict.txt
Normal file
153
ppocr/utils/dict/ka_dict.txt
Normal file
@@ -0,0 +1,153 @@
|
||||
k
|
||||
a
|
||||
_
|
||||
i
|
||||
m
|
||||
g
|
||||
/
|
||||
1
|
||||
2
|
||||
I
|
||||
L
|
||||
S
|
||||
V
|
||||
R
|
||||
C
|
||||
0
|
||||
v
|
||||
l
|
||||
6
|
||||
4
|
||||
8
|
||||
.
|
||||
j
|
||||
p
|
||||
ಗ
|
||||
ು
|
||||
ಣ
|
||||
ಪ
|
||||
ಡ
|
||||
ಿ
|
||||
ಸ
|
||||
ಲ
|
||||
ಾ
|
||||
ದ
|
||||
್
|
||||
7
|
||||
5
|
||||
3
|
||||
ವ
|
||||
ಷ
|
||||
ಬ
|
||||
ಹ
|
||||
ೆ
|
||||
9
|
||||
ಅ
|
||||
ಳ
|
||||
ನ
|
||||
ರ
|
||||
ಉ
|
||||
ಕ
|
||||
ಎ
|
||||
ೇ
|
||||
ಂ
|
||||
ೈ
|
||||
ೊ
|
||||
ೀ
|
||||
ಯ
|
||||
ೋ
|
||||
ತ
|
||||
ಶ
|
||||
ಭ
|
||||
ಧ
|
||||
ಚ
|
||||
ಜ
|
||||
ೂ
|
||||
ಮ
|
||||
ಒ
|
||||
ೃ
|
||||
ಥ
|
||||
ಇ
|
||||
ಟ
|
||||
ಖ
|
||||
ಆ
|
||||
ಞ
|
||||
ಫ
|
||||
-
|
||||
ಢ
|
||||
ಊ
|
||||
ಓ
|
||||
ಐ
|
||||
ಃ
|
||||
ಘ
|
||||
ಝ
|
||||
ೌ
|
||||
ಠ
|
||||
ಛ
|
||||
ಔ
|
||||
ಏ
|
||||
ಈ
|
||||
ಋ
|
||||
೨
|
||||
೦
|
||||
೧
|
||||
೮
|
||||
೯
|
||||
೪
|
||||
,
|
||||
೫
|
||||
೭
|
||||
೩
|
||||
೬
|
||||
ಙ
|
||||
s
|
||||
c
|
||||
e
|
||||
n
|
||||
w
|
||||
o
|
||||
u
|
||||
t
|
||||
d
|
||||
E
|
||||
A
|
||||
T
|
||||
B
|
||||
Z
|
||||
N
|
||||
G
|
||||
O
|
||||
q
|
||||
z
|
||||
r
|
||||
x
|
||||
P
|
||||
K
|
||||
M
|
||||
J
|
||||
U
|
||||
D
|
||||
f
|
||||
F
|
||||
h
|
||||
b
|
||||
W
|
||||
Y
|
||||
y
|
||||
H
|
||||
X
|
||||
Q
|
||||
'
|
||||
#
|
||||
&
|
||||
!
|
||||
@
|
||||
$
|
||||
:
|
||||
%
|
||||
é
|
||||
É
|
||||
(
|
||||
?
|
||||
+
|
||||
|
||||
42
ppocr/utils/dict/kazakh_dict.txt
Normal file
42
ppocr/utils/dict/kazakh_dict.txt
Normal file
@@ -0,0 +1,42 @@
|
||||
А
|
||||
Ә
|
||||
Б
|
||||
В
|
||||
Г
|
||||
Ғ
|
||||
Д
|
||||
Е
|
||||
Ё
|
||||
Ж
|
||||
З
|
||||
И
|
||||
Й
|
||||
К
|
||||
Қ
|
||||
Л
|
||||
М
|
||||
Н
|
||||
Ң
|
||||
О
|
||||
Ө
|
||||
П
|
||||
Р
|
||||
С
|
||||
Т
|
||||
У
|
||||
Ұ
|
||||
Ү
|
||||
Ф
|
||||
Х
|
||||
Һ
|
||||
Ц
|
||||
Ч
|
||||
Ш
|
||||
Щ
|
||||
Ъ
|
||||
Ы
|
||||
І
|
||||
Ь
|
||||
Э
|
||||
Ю
|
||||
Я
|
||||
4
ppocr/utils/dict/kie_dict/xfund_class_list.txt
Normal file
4
ppocr/utils/dict/kie_dict/xfund_class_list.txt
Normal file
@@ -0,0 +1,4 @@
|
||||
OTHER
|
||||
QUESTION
|
||||
ANSWER
|
||||
HEADER
|
||||
3688
ppocr/utils/dict/korean_dict.txt
Normal file
3688
ppocr/utils/dict/korean_dict.txt
Normal file
File diff suppressed because it is too large
Load Diff
1
ppocr/utils/dict/latex_ocr_tokenizer.json
Normal file
1
ppocr/utils/dict/latex_ocr_tokenizer.json
Normal file
File diff suppressed because one or more lines are too long
111
ppocr/utils/dict/latex_symbol_dict.txt
Normal file
111
ppocr/utils/dict/latex_symbol_dict.txt
Normal file
@@ -0,0 +1,111 @@
|
||||
eos
|
||||
sos
|
||||
!
|
||||
'
|
||||
(
|
||||
)
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
<
|
||||
=
|
||||
>
|
||||
A
|
||||
B
|
||||
C
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
L
|
||||
M
|
||||
N
|
||||
P
|
||||
R
|
||||
S
|
||||
T
|
||||
V
|
||||
X
|
||||
Y
|
||||
[
|
||||
\Delta
|
||||
\alpha
|
||||
\beta
|
||||
\cdot
|
||||
\cdots
|
||||
\cos
|
||||
\div
|
||||
\exists
|
||||
\forall
|
||||
\frac
|
||||
\gamma
|
||||
\geq
|
||||
\in
|
||||
\infty
|
||||
\int
|
||||
\lambda
|
||||
\ldots
|
||||
\leq
|
||||
\lim
|
||||
\log
|
||||
\mu
|
||||
\neq
|
||||
\phi
|
||||
\pi
|
||||
\pm
|
||||
\prime
|
||||
\rightarrow
|
||||
\sigma
|
||||
\sin
|
||||
\sqrt
|
||||
\sum
|
||||
\tan
|
||||
\theta
|
||||
\times
|
||||
]
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
\{
|
||||
|
|
||||
\}
|
||||
{
|
||||
}
|
||||
^
|
||||
_
|
||||
185
ppocr/utils/dict/latin_dict.txt
Normal file
185
ppocr/utils/dict/latin_dict.txt
Normal file
@@ -0,0 +1,185 @@
|
||||
|
||||
!
|
||||
"
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
)
|
||||
*
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
:
|
||||
;
|
||||
<
|
||||
=
|
||||
>
|
||||
?
|
||||
@
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
[
|
||||
]
|
||||
_
|
||||
`
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
{
|
||||
}
|
||||
¡
|
||||
£
|
||||
§
|
||||
ª
|
||||
«
|
||||
|
||||
°
|
||||
²
|
||||
³
|
||||
´
|
||||
µ
|
||||
·
|
||||
º
|
||||
»
|
||||
¿
|
||||
À
|
||||
Á
|
||||
Â
|
||||
Ä
|
||||
Å
|
||||
Ç
|
||||
È
|
||||
É
|
||||
Ê
|
||||
Ë
|
||||
Ì
|
||||
Í
|
||||
Î
|
||||
Ï
|
||||
Ò
|
||||
Ó
|
||||
Ô
|
||||
Õ
|
||||
Ö
|
||||
Ú
|
||||
Ü
|
||||
Ý
|
||||
ß
|
||||
à
|
||||
á
|
||||
â
|
||||
ã
|
||||
ä
|
||||
å
|
||||
æ
|
||||
ç
|
||||
è
|
||||
é
|
||||
ê
|
||||
ë
|
||||
ì
|
||||
í
|
||||
î
|
||||
ï
|
||||
ñ
|
||||
ò
|
||||
ó
|
||||
ô
|
||||
õ
|
||||
ö
|
||||
ø
|
||||
ù
|
||||
ú
|
||||
û
|
||||
ü
|
||||
ý
|
||||
ą
|
||||
Ć
|
||||
ć
|
||||
Č
|
||||
č
|
||||
Đ
|
||||
đ
|
||||
ę
|
||||
ı
|
||||
Ł
|
||||
ł
|
||||
ō
|
||||
Œ
|
||||
œ
|
||||
Š
|
||||
š
|
||||
Ÿ
|
||||
Ž
|
||||
ž
|
||||
ʒ
|
||||
β
|
||||
δ
|
||||
ε
|
||||
з
|
||||
Ṡ
|
||||
‘
|
||||
€
|
||||
™
|
||||
10
ppocr/utils/dict/layout_dict/layout_cdla_dict.txt
Normal file
10
ppocr/utils/dict/layout_dict/layout_cdla_dict.txt
Normal file
@@ -0,0 +1,10 @@
|
||||
text
|
||||
title
|
||||
figure
|
||||
figure_caption
|
||||
table
|
||||
table_caption
|
||||
header
|
||||
footer
|
||||
reference
|
||||
equation
|
||||
5
ppocr/utils/dict/layout_dict/layout_publaynet_dict.txt
Normal file
5
ppocr/utils/dict/layout_dict/layout_publaynet_dict.txt
Normal file
@@ -0,0 +1,5 @@
|
||||
text
|
||||
title
|
||||
list
|
||||
table
|
||||
figure
|
||||
1
ppocr/utils/dict/layout_dict/layout_table_dict.txt
Normal file
1
ppocr/utils/dict/layout_dict/layout_table_dict.txt
Normal file
@@ -0,0 +1 @@
|
||||
table
|
||||
153
ppocr/utils/dict/mr_dict.txt
Normal file
153
ppocr/utils/dict/mr_dict.txt
Normal file
@@ -0,0 +1,153 @@
|
||||
|
||||
!
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
:
|
||||
?
|
||||
@
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
_
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
É
|
||||
é
|
||||
ँ
|
||||
ं
|
||||
ः
|
||||
अ
|
||||
आ
|
||||
इ
|
||||
ई
|
||||
उ
|
||||
ऊ
|
||||
ए
|
||||
ऐ
|
||||
ऑ
|
||||
ओ
|
||||
औ
|
||||
क
|
||||
ख
|
||||
ग
|
||||
घ
|
||||
च
|
||||
छ
|
||||
ज
|
||||
झ
|
||||
ञ
|
||||
ट
|
||||
ठ
|
||||
ड
|
||||
ढ
|
||||
ण
|
||||
त
|
||||
थ
|
||||
द
|
||||
ध
|
||||
न
|
||||
प
|
||||
फ
|
||||
ब
|
||||
भ
|
||||
म
|
||||
य
|
||||
र
|
||||
ऱ
|
||||
ल
|
||||
ळ
|
||||
व
|
||||
श
|
||||
ष
|
||||
स
|
||||
ह
|
||||
़
|
||||
ा
|
||||
ि
|
||||
ी
|
||||
ु
|
||||
ू
|
||||
ृ
|
||||
ॅ
|
||||
े
|
||||
ै
|
||||
ॉ
|
||||
ो
|
||||
ौ
|
||||
्
|
||||
०
|
||||
१
|
||||
२
|
||||
३
|
||||
४
|
||||
५
|
||||
६
|
||||
७
|
||||
८
|
||||
९
|
||||
153
ppocr/utils/dict/ne_dict.txt
Normal file
153
ppocr/utils/dict/ne_dict.txt
Normal file
@@ -0,0 +1,153 @@
|
||||
|
||||
!
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
:
|
||||
?
|
||||
@
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
_
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
É
|
||||
é
|
||||
ः
|
||||
अ
|
||||
आ
|
||||
इ
|
||||
ई
|
||||
उ
|
||||
ऊ
|
||||
ऋ
|
||||
ए
|
||||
ऐ
|
||||
ओ
|
||||
औ
|
||||
क
|
||||
ख
|
||||
ग
|
||||
घ
|
||||
ङ
|
||||
च
|
||||
छ
|
||||
ज
|
||||
झ
|
||||
ञ
|
||||
ट
|
||||
ठ
|
||||
ड
|
||||
ढ
|
||||
ण
|
||||
त
|
||||
थ
|
||||
द
|
||||
ध
|
||||
न
|
||||
ऩ
|
||||
प
|
||||
फ
|
||||
ब
|
||||
भ
|
||||
म
|
||||
य
|
||||
र
|
||||
ऱ
|
||||
ल
|
||||
व
|
||||
श
|
||||
ष
|
||||
स
|
||||
ह
|
||||
़
|
||||
ा
|
||||
ि
|
||||
ी
|
||||
ु
|
||||
ू
|
||||
ृ
|
||||
े
|
||||
ै
|
||||
ो
|
||||
ौ
|
||||
्
|
||||
॒
|
||||
ॠ
|
||||
।
|
||||
०
|
||||
१
|
||||
२
|
||||
३
|
||||
४
|
||||
५
|
||||
६
|
||||
७
|
||||
८
|
||||
९
|
||||
96
ppocr/utils/dict/oc_dict.txt
Normal file
96
ppocr/utils/dict/oc_dict.txt
Normal file
@@ -0,0 +1,96 @@
|
||||
o
|
||||
c
|
||||
_
|
||||
i
|
||||
m
|
||||
g
|
||||
/
|
||||
2
|
||||
0
|
||||
I
|
||||
L
|
||||
S
|
||||
V
|
||||
R
|
||||
C
|
||||
1
|
||||
v
|
||||
a
|
||||
l
|
||||
4
|
||||
3
|
||||
.
|
||||
j
|
||||
p
|
||||
r
|
||||
e
|
||||
è
|
||||
t
|
||||
9
|
||||
7
|
||||
5
|
||||
8
|
||||
n
|
||||
'
|
||||
b
|
||||
s
|
||||
6
|
||||
q
|
||||
u
|
||||
á
|
||||
d
|
||||
ò
|
||||
à
|
||||
h
|
||||
z
|
||||
f
|
||||
ï
|
||||
í
|
||||
A
|
||||
ç
|
||||
x
|
||||
ó
|
||||
é
|
||||
P
|
||||
O
|
||||
Ò
|
||||
ü
|
||||
k
|
||||
À
|
||||
F
|
||||
-
|
||||
ú
|
||||
|
||||
æ
|
||||
Á
|
||||
D
|
||||
E
|
||||
w
|
||||
K
|
||||
T
|
||||
N
|
||||
y
|
||||
U
|
||||
Z
|
||||
G
|
||||
B
|
||||
J
|
||||
H
|
||||
M
|
||||
W
|
||||
Y
|
||||
X
|
||||
Q
|
||||
%
|
||||
$
|
||||
,
|
||||
@
|
||||
&
|
||||
!
|
||||
:
|
||||
(
|
||||
#
|
||||
?
|
||||
+
|
||||
É
|
||||
|
||||
94
ppocr/utils/dict/parseq_dict.txt
Normal file
94
ppocr/utils/dict/parseq_dict.txt
Normal file
@@ -0,0 +1,94 @@
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
!
|
||||
"
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
)
|
||||
*
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
:
|
||||
;
|
||||
<
|
||||
=
|
||||
>
|
||||
?
|
||||
@
|
||||
[
|
||||
\
|
||||
]
|
||||
^
|
||||
_
|
||||
`
|
||||
{
|
||||
|
|
||||
}
|
||||
~
|
||||
15629
ppocr/utils/dict/ppocrv4_doc_dict.txt
Normal file
15629
ppocr/utils/dict/ppocrv4_doc_dict.txt
Normal file
File diff suppressed because it is too large
Load Diff
18383
ppocr/utils/dict/ppocrv5_dict.txt
Normal file
18383
ppocr/utils/dict/ppocrv5_dict.txt
Normal file
File diff suppressed because it is too large
Load Diff
517
ppocr/utils/dict/ppocrv5_eslav_dict.txt
Normal file
517
ppocr/utils/dict/ppocrv5_eslav_dict.txt
Normal file
@@ -0,0 +1,517 @@
|
||||
!
|
||||
"
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
)
|
||||
*
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
:
|
||||
;
|
||||
<
|
||||
=
|
||||
>
|
||||
?
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
[
|
||||
]
|
||||
_
|
||||
`
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
©
|
||||
‥
|
||||
{
|
||||
}
|
||||
\
|
||||
|
|
||||
@
|
||||
^
|
||||
~
|
||||
÷
|
||||
∕
|
||||
∙
|
||||
⋅
|
||||
·
|
||||
±
|
||||
∓
|
||||
∩
|
||||
∪
|
||||
□
|
||||
←
|
||||
↔
|
||||
⇒
|
||||
⇐
|
||||
⇔
|
||||
∀
|
||||
∃
|
||||
∄
|
||||
∴
|
||||
∵
|
||||
∝
|
||||
∞
|
||||
⊥
|
||||
∟
|
||||
∠
|
||||
∡
|
||||
∢
|
||||
′
|
||||
″
|
||||
∥
|
||||
⊾
|
||||
⊿
|
||||
∂
|
||||
∫
|
||||
∬
|
||||
∭
|
||||
∮
|
||||
∯
|
||||
∰
|
||||
∑
|
||||
∏
|
||||
√
|
||||
∛
|
||||
∜
|
||||
∱
|
||||
∲
|
||||
∳
|
||||
∶
|
||||
∷
|
||||
∼
|
||||
®
|
||||
℉
|
||||
Ω
|
||||
℧
|
||||
Å
|
||||
⌀
|
||||
ℏ
|
||||
⅀
|
||||
⍺
|
||||
⍵
|
||||
¢
|
||||
€
|
||||
£
|
||||
¥
|
||||
₿
|
||||
Ⅰ
|
||||
Ⅱ
|
||||
Ⅲ
|
||||
Ⅳ
|
||||
Ⅴ
|
||||
Ⅵ
|
||||
Ⅶ
|
||||
Ⅷ
|
||||
Ⅸ
|
||||
Ⅹ
|
||||
Ⅺ
|
||||
Ⅻ
|
||||
ⅰ
|
||||
ⅱ
|
||||
ⅲ
|
||||
ⅳ
|
||||
ⅴ
|
||||
ⅵ
|
||||
ⅶ
|
||||
ⅷ
|
||||
ⅸ
|
||||
ⅹ
|
||||
ⅺ
|
||||
ⅻ
|
||||
➀
|
||||
➁
|
||||
➂
|
||||
➃
|
||||
➄
|
||||
➅
|
||||
➆
|
||||
➇
|
||||
➈
|
||||
➉
|
||||
➊
|
||||
➋
|
||||
➌
|
||||
➍
|
||||
➎
|
||||
➏
|
||||
➐
|
||||
➑
|
||||
➒
|
||||
➓
|
||||
❶
|
||||
❷
|
||||
❸
|
||||
❹
|
||||
❺
|
||||
❻
|
||||
❼
|
||||
❽
|
||||
❾
|
||||
❿
|
||||
①
|
||||
②
|
||||
③
|
||||
④
|
||||
⑤
|
||||
⑥
|
||||
⑦
|
||||
⑧
|
||||
⑨
|
||||
⑩
|
||||
●
|
||||
▶
|
||||
𝑢
|
||||
︽
|
||||
–
|
||||
﹥
|
||||
𝜓
|
||||
•
|
||||
∋
|
||||
ƒ
|
||||
०
|
||||
⬆
|
||||
Ạ
|
||||
◀
|
||||
|
||||
▫
|
||||
︾
|
||||
À
|
||||
Á
|
||||
Â
|
||||
Ã
|
||||
Ä
|
||||
Å
|
||||
Æ
|
||||
Ç
|
||||
È
|
||||
É
|
||||
Ê
|
||||
Ë
|
||||
Ì
|
||||
Í
|
||||
Î
|
||||
Ï
|
||||
Ð
|
||||
Ñ
|
||||
Ò
|
||||
Ó
|
||||
Ô
|
||||
Õ
|
||||
Ö
|
||||
Ø
|
||||
Ù
|
||||
Ú
|
||||
Û
|
||||
Ü
|
||||
Ý
|
||||
Þ
|
||||
à
|
||||
á
|
||||
â
|
||||
ã
|
||||
ä
|
||||
å
|
||||
æ
|
||||
ç
|
||||
è
|
||||
é
|
||||
ê
|
||||
ë
|
||||
ì
|
||||
í
|
||||
î
|
||||
ï
|
||||
ð
|
||||
ñ
|
||||
ò
|
||||
ó
|
||||
ô
|
||||
õ
|
||||
ö
|
||||
ø
|
||||
ù
|
||||
ú
|
||||
û
|
||||
ü
|
||||
ý
|
||||
þ
|
||||
ÿ
|
||||
¡
|
||||
¤
|
||||
¦
|
||||
§
|
||||
¨
|
||||
ª
|
||||
«
|
||||
¬
|
||||
¯
|
||||
°
|
||||
²
|
||||
³
|
||||
´
|
||||
µ
|
||||
¶
|
||||
¸
|
||||
¹
|
||||
º
|
||||
»
|
||||
¼
|
||||
½
|
||||
¾
|
||||
¿
|
||||
×
|
||||
‐
|
||||
‑
|
||||
‒
|
||||
—
|
||||
―
|
||||
‖
|
||||
‗
|
||||
‘
|
||||
’
|
||||
‚
|
||||
‛
|
||||
“
|
||||
”
|
||||
„
|
||||
‟
|
||||
†
|
||||
‡
|
||||
‣
|
||||
․
|
||||
…
|
||||
‧
|
||||
‰
|
||||
‴
|
||||
‵
|
||||
‶
|
||||
‷
|
||||
‸
|
||||
‹
|
||||
›
|
||||
※
|
||||
‼
|
||||
‽
|
||||
‾
|
||||
₤
|
||||
₡
|
||||
₹
|
||||
−
|
||||
∖
|
||||
∗
|
||||
≈
|
||||
≠
|
||||
≡
|
||||
≤
|
||||
≥
|
||||
⊂
|
||||
⊃
|
||||
↑
|
||||
→
|
||||
↓
|
||||
↕
|
||||
™
|
||||
Ω
|
||||
℮
|
||||
∆
|
||||
✓
|
||||
✗
|
||||
✘
|
||||
▪
|
||||
◼
|
||||
✔
|
||||
✕
|
||||
☑
|
||||
☒
|
||||
№
|
||||
₽
|
||||
₴
|
||||
Α
|
||||
α
|
||||
Β
|
||||
β
|
||||
Γ
|
||||
γ
|
||||
Δ
|
||||
δ
|
||||
Ε
|
||||
ε
|
||||
Ζ
|
||||
ζ
|
||||
Η
|
||||
η
|
||||
Θ
|
||||
θ
|
||||
Ι
|
||||
ι
|
||||
Κ
|
||||
κ
|
||||
Λ
|
||||
λ
|
||||
Μ
|
||||
μ
|
||||
Ν
|
||||
ν
|
||||
Ξ
|
||||
ξ
|
||||
Ο
|
||||
ο
|
||||
Π
|
||||
π
|
||||
Ρ
|
||||
ρ
|
||||
Σ
|
||||
σ
|
||||
ς
|
||||
Τ
|
||||
τ
|
||||
Υ
|
||||
υ
|
||||
Φ
|
||||
φ
|
||||
Χ
|
||||
χ
|
||||
Ψ
|
||||
ψ
|
||||
ω
|
||||
А
|
||||
Б
|
||||
В
|
||||
Г
|
||||
Ґ
|
||||
Д
|
||||
Е
|
||||
Ё
|
||||
Є
|
||||
Ж
|
||||
З
|
||||
И
|
||||
І
|
||||
Ї
|
||||
Й
|
||||
К
|
||||
Л
|
||||
М
|
||||
Н
|
||||
О
|
||||
П
|
||||
Р
|
||||
С
|
||||
Т
|
||||
У
|
||||
Ў
|
||||
Ф
|
||||
Х
|
||||
Ц
|
||||
Ч
|
||||
Ш
|
||||
Щ
|
||||
Ъ
|
||||
Ы
|
||||
Ь
|
||||
Э
|
||||
Ю
|
||||
Я
|
||||
а
|
||||
б
|
||||
в
|
||||
г
|
||||
ґ
|
||||
д
|
||||
е
|
||||
ё
|
||||
є
|
||||
ж
|
||||
з
|
||||
и
|
||||
і
|
||||
ї
|
||||
й
|
||||
к
|
||||
л
|
||||
м
|
||||
н
|
||||
о
|
||||
п
|
||||
р
|
||||
с
|
||||
т
|
||||
у
|
||||
ў
|
||||
ф
|
||||
х
|
||||
ц
|
||||
ч
|
||||
ш
|
||||
щ
|
||||
ъ
|
||||
ы
|
||||
ь
|
||||
э
|
||||
ю
|
||||
я
|
||||
11945
ppocr/utils/dict/ppocrv5_korean_dict.txt
Normal file
11945
ppocr/utils/dict/ppocrv5_korean_dict.txt
Normal file
File diff suppressed because it is too large
Load Diff
502
ppocr/utils/dict/ppocrv5_latin_dict.txt
Normal file
502
ppocr/utils/dict/ppocrv5_latin_dict.txt
Normal file
@@ -0,0 +1,502 @@
|
||||
!
|
||||
"
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
)
|
||||
*
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
:
|
||||
;
|
||||
<
|
||||
=
|
||||
>
|
||||
?
|
||||
@
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
[
|
||||
\
|
||||
]
|
||||
^
|
||||
_
|
||||
`
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
{
|
||||
|
|
||||
}
|
||||
~
|
||||
¡
|
||||
¢
|
||||
£
|
||||
¤
|
||||
¥
|
||||
¦
|
||||
§
|
||||
¨
|
||||
©
|
||||
ª
|
||||
«
|
||||
¬
|
||||
|
||||
®
|
||||
¯
|
||||
°
|
||||
±
|
||||
²
|
||||
³
|
||||
´
|
||||
µ
|
||||
¶
|
||||
·
|
||||
¸
|
||||
¹
|
||||
º
|
||||
»
|
||||
¼
|
||||
½
|
||||
¾
|
||||
¿
|
||||
À
|
||||
Á
|
||||
Â
|
||||
Ã
|
||||
Ä
|
||||
Å
|
||||
Æ
|
||||
Ç
|
||||
È
|
||||
É
|
||||
Ê
|
||||
Ë
|
||||
Ì
|
||||
Í
|
||||
Î
|
||||
Ï
|
||||
Ð
|
||||
Ñ
|
||||
Ò
|
||||
Ó
|
||||
Ô
|
||||
Õ
|
||||
Ö
|
||||
×
|
||||
Ø
|
||||
Ù
|
||||
Ú
|
||||
Û
|
||||
Ü
|
||||
Ý
|
||||
Þ
|
||||
ß
|
||||
à
|
||||
á
|
||||
â
|
||||
ã
|
||||
ä
|
||||
å
|
||||
æ
|
||||
ç
|
||||
è
|
||||
é
|
||||
ê
|
||||
ë
|
||||
ì
|
||||
í
|
||||
î
|
||||
ï
|
||||
ð
|
||||
ñ
|
||||
ò
|
||||
ó
|
||||
ô
|
||||
õ
|
||||
ö
|
||||
÷
|
||||
ø
|
||||
ù
|
||||
ú
|
||||
û
|
||||
ü
|
||||
ý
|
||||
þ
|
||||
ÿ
|
||||
Ą
|
||||
ą
|
||||
Ć
|
||||
ć
|
||||
Č
|
||||
č
|
||||
Ď
|
||||
ď
|
||||
Đ
|
||||
đ
|
||||
Ė
|
||||
ė
|
||||
Ę
|
||||
ę
|
||||
Ě
|
||||
ě
|
||||
Ğ
|
||||
ğ
|
||||
Į
|
||||
į
|
||||
İ
|
||||
ı
|
||||
Ĺ
|
||||
ĺ
|
||||
Ľ
|
||||
ľ
|
||||
Ł
|
||||
ł
|
||||
Ń
|
||||
ń
|
||||
Ň
|
||||
ň
|
||||
ō
|
||||
Ő
|
||||
ő
|
||||
Œ
|
||||
œ
|
||||
Ŕ
|
||||
ŕ
|
||||
Ř
|
||||
ř
|
||||
Ś
|
||||
ś
|
||||
Ş
|
||||
ş
|
||||
Š
|
||||
š
|
||||
Ť
|
||||
ť
|
||||
Ū
|
||||
ū
|
||||
Ů
|
||||
ů
|
||||
Ű
|
||||
ű
|
||||
Ų
|
||||
ų
|
||||
Ÿ
|
||||
Ź
|
||||
ź
|
||||
Ż
|
||||
ż
|
||||
Ž
|
||||
ž
|
||||
ƒ
|
||||
ʒ
|
||||
Ω
|
||||
α
|
||||
β
|
||||
γ
|
||||
δ
|
||||
ε
|
||||
ζ
|
||||
η
|
||||
θ
|
||||
ι
|
||||
κ
|
||||
λ
|
||||
μ
|
||||
ν
|
||||
ξ
|
||||
ο
|
||||
π
|
||||
ρ
|
||||
ς
|
||||
σ
|
||||
τ
|
||||
υ
|
||||
φ
|
||||
χ
|
||||
ψ
|
||||
ω
|
||||
з
|
||||
०
|
||||
Ṡ
|
||||
ẞ
|
||||
Ạ
|
||||
‐
|
||||
‑
|
||||
‒
|
||||
–
|
||||
—
|
||||
―
|
||||
‖
|
||||
‗
|
||||
‘
|
||||
’
|
||||
‚
|
||||
‛
|
||||
“
|
||||
”
|
||||
„
|
||||
‟
|
||||
†
|
||||
‡
|
||||
•
|
||||
‣
|
||||
․
|
||||
‥
|
||||
…
|
||||
‧
|
||||
‰
|
||||
′
|
||||
″
|
||||
‴
|
||||
‵
|
||||
‶
|
||||
‷
|
||||
‸
|
||||
‹
|
||||
›
|
||||
※
|
||||
‼
|
||||
‽
|
||||
‾
|
||||
⁄
|
||||
₂
|
||||
₃
|
||||
₡
|
||||
₤
|
||||
€
|
||||
₴
|
||||
₹
|
||||
₽
|
||||
₿
|
||||
℉
|
||||
ℏ
|
||||
№
|
||||
™
|
||||
Ω
|
||||
℧
|
||||
Å
|
||||
℮
|
||||
⅀
|
||||
Ⅰ
|
||||
Ⅱ
|
||||
Ⅲ
|
||||
Ⅳ
|
||||
Ⅴ
|
||||
Ⅵ
|
||||
Ⅶ
|
||||
Ⅷ
|
||||
Ⅸ
|
||||
Ⅹ
|
||||
Ⅺ
|
||||
Ⅻ
|
||||
ⅰ
|
||||
ⅱ
|
||||
ⅲ
|
||||
ⅳ
|
||||
ⅴ
|
||||
ⅵ
|
||||
ⅶ
|
||||
ⅷ
|
||||
ⅸ
|
||||
ⅹ
|
||||
ⅺ
|
||||
ⅻ
|
||||
←
|
||||
↑
|
||||
→
|
||||
↓
|
||||
↔
|
||||
↕
|
||||
⇐
|
||||
⇒
|
||||
⇔
|
||||
∀
|
||||
∂
|
||||
∃
|
||||
∄
|
||||
∅
|
||||
∆
|
||||
∋
|
||||
∏
|
||||
∑
|
||||
−
|
||||
∓
|
||||
∕
|
||||
∖
|
||||
∗
|
||||
∙
|
||||
√
|
||||
∛
|
||||
∜
|
||||
∝
|
||||
∞
|
||||
∟
|
||||
∠
|
||||
∡
|
||||
∢
|
||||
∥
|
||||
∧
|
||||
∨
|
||||
∩
|
||||
∪
|
||||
∫
|
||||
∬
|
||||
∭
|
||||
∮
|
||||
∯
|
||||
∰
|
||||
∱
|
||||
∲
|
||||
∳
|
||||
∴
|
||||
∵
|
||||
∶
|
||||
∷
|
||||
∼
|
||||
≈
|
||||
≠
|
||||
≡
|
||||
≤
|
||||
≥
|
||||
⊂
|
||||
⊃
|
||||
⊥
|
||||
⊾
|
||||
⊿
|
||||
⋅
|
||||
⌀
|
||||
⍵
|
||||
⍺
|
||||
①
|
||||
②
|
||||
③
|
||||
④
|
||||
⑤
|
||||
⑥
|
||||
⑦
|
||||
⑧
|
||||
⑨
|
||||
⑩
|
||||
─
|
||||
│
|
||||
└
|
||||
├
|
||||
■
|
||||
□
|
||||
▪
|
||||
▫
|
||||
▶
|
||||
◀
|
||||
●
|
||||
◼
|
||||
☑
|
||||
☒
|
||||
✓
|
||||
✔
|
||||
✕
|
||||
✗
|
||||
✘
|
||||
❶
|
||||
❷
|
||||
❸
|
||||
❹
|
||||
❺
|
||||
❻
|
||||
❼
|
||||
❽
|
||||
❾
|
||||
❿
|
||||
➀
|
||||
➁
|
||||
➂
|
||||
➃
|
||||
➄
|
||||
➅
|
||||
➆
|
||||
➇
|
||||
➈
|
||||
➉
|
||||
➊
|
||||
➋
|
||||
➌
|
||||
➍
|
||||
➎
|
||||
➏
|
||||
➐
|
||||
➑
|
||||
➒
|
||||
➓
|
||||
⬆
|
||||
、
|
||||
fi
|
||||
fl
|
||||
︽
|
||||
︾
|
||||
﹥
|
||||
<EFBFBD>
|
||||
𝑢
|
||||
𝜓
|
||||
130
ppocr/utils/dict/pu_dict.txt
Normal file
130
ppocr/utils/dict/pu_dict.txt
Normal file
@@ -0,0 +1,130 @@
|
||||
p
|
||||
u
|
||||
_
|
||||
i
|
||||
m
|
||||
g
|
||||
/
|
||||
8
|
||||
I
|
||||
L
|
||||
S
|
||||
V
|
||||
R
|
||||
C
|
||||
2
|
||||
0
|
||||
1
|
||||
v
|
||||
a
|
||||
l
|
||||
6
|
||||
7
|
||||
4
|
||||
5
|
||||
.
|
||||
j
|
||||
|
||||
q
|
||||
e
|
||||
s
|
||||
t
|
||||
ã
|
||||
o
|
||||
x
|
||||
9
|
||||
c
|
||||
n
|
||||
r
|
||||
z
|
||||
ç
|
||||
õ
|
||||
3
|
||||
A
|
||||
U
|
||||
d
|
||||
º
|
||||
ô
|
||||
|
||||
,
|
||||
E
|
||||
;
|
||||
ó
|
||||
á
|
||||
b
|
||||
D
|
||||
?
|
||||
ú
|
||||
ê
|
||||
-
|
||||
h
|
||||
P
|
||||
f
|
||||
à
|
||||
N
|
||||
í
|
||||
O
|
||||
M
|
||||
G
|
||||
É
|
||||
é
|
||||
â
|
||||
F
|
||||
:
|
||||
T
|
||||
Á
|
||||
"
|
||||
Q
|
||||
)
|
||||
W
|
||||
J
|
||||
B
|
||||
H
|
||||
(
|
||||
ö
|
||||
%
|
||||
Ö
|
||||
«
|
||||
w
|
||||
K
|
||||
y
|
||||
!
|
||||
k
|
||||
]
|
||||
'
|
||||
Z
|
||||
+
|
||||
Ç
|
||||
Õ
|
||||
Y
|
||||
À
|
||||
X
|
||||
µ
|
||||
»
|
||||
ª
|
||||
Í
|
||||
ü
|
||||
ä
|
||||
´
|
||||
è
|
||||
ñ
|
||||
ß
|
||||
ï
|
||||
Ú
|
||||
ë
|
||||
Ô
|
||||
Ï
|
||||
Ó
|
||||
[
|
||||
Ì
|
||||
<
|
||||
Â
|
||||
ò
|
||||
§
|
||||
³
|
||||
ø
|
||||
å
|
||||
#
|
||||
$
|
||||
&
|
||||
@
|
||||
91
ppocr/utils/dict/rs_dict.txt
Normal file
91
ppocr/utils/dict/rs_dict.txt
Normal file
@@ -0,0 +1,91 @@
|
||||
r
|
||||
s
|
||||
_
|
||||
i
|
||||
m
|
||||
g
|
||||
/
|
||||
1
|
||||
I
|
||||
L
|
||||
S
|
||||
V
|
||||
R
|
||||
C
|
||||
2
|
||||
0
|
||||
v
|
||||
a
|
||||
l
|
||||
7
|
||||
5
|
||||
8
|
||||
6
|
||||
.
|
||||
j
|
||||
p
|
||||
|
||||
t
|
||||
d
|
||||
9
|
||||
3
|
||||
e
|
||||
š
|
||||
4
|
||||
k
|
||||
u
|
||||
ć
|
||||
c
|
||||
n
|
||||
đ
|
||||
o
|
||||
z
|
||||
č
|
||||
b
|
||||
ž
|
||||
f
|
||||
Z
|
||||
T
|
||||
h
|
||||
M
|
||||
F
|
||||
O
|
||||
Š
|
||||
B
|
||||
H
|
||||
A
|
||||
E
|
||||
Đ
|
||||
Ž
|
||||
D
|
||||
P
|
||||
G
|
||||
Č
|
||||
K
|
||||
U
|
||||
N
|
||||
J
|
||||
Ć
|
||||
w
|
||||
y
|
||||
W
|
||||
x
|
||||
Y
|
||||
X
|
||||
q
|
||||
Q
|
||||
#
|
||||
&
|
||||
$
|
||||
,
|
||||
-
|
||||
%
|
||||
'
|
||||
@
|
||||
!
|
||||
:
|
||||
?
|
||||
(
|
||||
É
|
||||
é
|
||||
+
|
||||
134
ppocr/utils/dict/rsc_dict.txt
Normal file
134
ppocr/utils/dict/rsc_dict.txt
Normal file
@@ -0,0 +1,134 @@
|
||||
r
|
||||
s
|
||||
c
|
||||
_
|
||||
i
|
||||
m
|
||||
g
|
||||
/
|
||||
5
|
||||
I
|
||||
L
|
||||
S
|
||||
V
|
||||
R
|
||||
C
|
||||
2
|
||||
0
|
||||
1
|
||||
v
|
||||
a
|
||||
l
|
||||
9
|
||||
7
|
||||
8
|
||||
.
|
||||
j
|
||||
p
|
||||
м
|
||||
а
|
||||
с
|
||||
и
|
||||
р
|
||||
ћ
|
||||
е
|
||||
ш
|
||||
3
|
||||
4
|
||||
о
|
||||
г
|
||||
н
|
||||
з
|
||||
в
|
||||
л
|
||||
6
|
||||
т
|
||||
ж
|
||||
у
|
||||
к
|
||||
п
|
||||
њ
|
||||
д
|
||||
ч
|
||||
С
|
||||
ј
|
||||
ф
|
||||
ц
|
||||
љ
|
||||
х
|
||||
О
|
||||
И
|
||||
А
|
||||
б
|
||||
Ш
|
||||
К
|
||||
ђ
|
||||
џ
|
||||
М
|
||||
В
|
||||
З
|
||||
Д
|
||||
Р
|
||||
У
|
||||
Н
|
||||
Т
|
||||
Б
|
||||
?
|
||||
П
|
||||
Х
|
||||
Ј
|
||||
Ц
|
||||
Г
|
||||
Љ
|
||||
Л
|
||||
Ф
|
||||
e
|
||||
n
|
||||
w
|
||||
E
|
||||
F
|
||||
A
|
||||
N
|
||||
f
|
||||
o
|
||||
b
|
||||
M
|
||||
G
|
||||
t
|
||||
y
|
||||
W
|
||||
k
|
||||
P
|
||||
u
|
||||
H
|
||||
B
|
||||
T
|
||||
z
|
||||
h
|
||||
O
|
||||
Y
|
||||
d
|
||||
U
|
||||
K
|
||||
D
|
||||
x
|
||||
X
|
||||
J
|
||||
Z
|
||||
Q
|
||||
q
|
||||
'
|
||||
-
|
||||
@
|
||||
é
|
||||
#
|
||||
!
|
||||
,
|
||||
%
|
||||
$
|
||||
:
|
||||
&
|
||||
+
|
||||
(
|
||||
É
|
||||
|
||||
125
ppocr/utils/dict/ru_dict.txt
Normal file
125
ppocr/utils/dict/ru_dict.txt
Normal file
@@ -0,0 +1,125 @@
|
||||
к
|
||||
в
|
||||
а
|
||||
з
|
||||
и
|
||||
у
|
||||
р
|
||||
о
|
||||
н
|
||||
я
|
||||
х
|
||||
п
|
||||
л
|
||||
ы
|
||||
г
|
||||
е
|
||||
т
|
||||
м
|
||||
д
|
||||
ж
|
||||
ш
|
||||
ь
|
||||
с
|
||||
ё
|
||||
б
|
||||
й
|
||||
ч
|
||||
ю
|
||||
ц
|
||||
щ
|
||||
М
|
||||
э
|
||||
ф
|
||||
А
|
||||
ъ
|
||||
С
|
||||
Ф
|
||||
Ю
|
||||
В
|
||||
К
|
||||
Т
|
||||
Н
|
||||
О
|
||||
Э
|
||||
У
|
||||
И
|
||||
Г
|
||||
Л
|
||||
Р
|
||||
Д
|
||||
Б
|
||||
Ш
|
||||
П
|
||||
З
|
||||
Х
|
||||
Е
|
||||
Ж
|
||||
Я
|
||||
Ц
|
||||
Ч
|
||||
Й
|
||||
Щ
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
|
||||
222
ppocr/utils/dict/samaritan_dict.txt
Normal file
222
ppocr/utils/dict/samaritan_dict.txt
Normal file
@@ -0,0 +1,222 @@
|
||||
!
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
:
|
||||
?
|
||||
@
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
_
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
É
|
||||
é
|
||||
ء
|
||||
آ
|
||||
أ
|
||||
ؤ
|
||||
إ
|
||||
ئ
|
||||
ا
|
||||
ب
|
||||
ة
|
||||
ت
|
||||
ث
|
||||
ج
|
||||
ح
|
||||
خ
|
||||
د
|
||||
ذ
|
||||
ر
|
||||
ز
|
||||
س
|
||||
ش
|
||||
ص
|
||||
ض
|
||||
ط
|
||||
ظ
|
||||
ع
|
||||
غ
|
||||
ف
|
||||
ق
|
||||
ك
|
||||
ل
|
||||
م
|
||||
ن
|
||||
ه
|
||||
و
|
||||
ى
|
||||
ي
|
||||
ً
|
||||
ٌ
|
||||
ٍ
|
||||
َ
|
||||
ُ
|
||||
ِ
|
||||
ّ
|
||||
ْ
|
||||
ٓ
|
||||
ٔ
|
||||
ٰ
|
||||
ٱ
|
||||
ٹ
|
||||
پ
|
||||
چ
|
||||
ڈ
|
||||
ڑ
|
||||
ژ
|
||||
ک
|
||||
ڭ
|
||||
گ
|
||||
ں
|
||||
ھ
|
||||
ۀ
|
||||
ہ
|
||||
ۂ
|
||||
ۃ
|
||||
ۆ
|
||||
ۇ
|
||||
ۈ
|
||||
ۋ
|
||||
ی
|
||||
ې
|
||||
ے
|
||||
ۓ
|
||||
ە
|
||||
١
|
||||
٢
|
||||
٣
|
||||
٤
|
||||
٥
|
||||
٦
|
||||
٧
|
||||
٨
|
||||
٩
|
||||
ࠀ
|
||||
ࠁ
|
||||
ࠂ
|
||||
ࠃ
|
||||
ࠄ
|
||||
ࠅ
|
||||
ࠆ
|
||||
ࠇ
|
||||
ࠈ
|
||||
ࠉ
|
||||
ࠊ
|
||||
ࠋ
|
||||
ࠌ
|
||||
ࠍ
|
||||
ࠎ
|
||||
ࠏ
|
||||
ࠐ
|
||||
ࠑ
|
||||
ࠒ
|
||||
ࠓ
|
||||
ࠔ
|
||||
ࠕ
|
||||
ࠖ
|
||||
ࠗ
|
||||
࠘
|
||||
࠙
|
||||
ࠚ
|
||||
ࠛ
|
||||
ࠜ
|
||||
ࠝ
|
||||
ࠞ
|
||||
ࠟ
|
||||
ࠠ
|
||||
ࠡ
|
||||
ࠢ
|
||||
ࠣ
|
||||
ࠤ
|
||||
ࠥ
|
||||
ࠦ
|
||||
ࠧ
|
||||
ࠨ
|
||||
ࠩ
|
||||
ࠪ
|
||||
ࠫ
|
||||
ࠬ
|
||||
࠭
|
||||
࠰
|
||||
࠱
|
||||
࠲
|
||||
࠳
|
||||
࠴
|
||||
࠵
|
||||
࠶
|
||||
࠷
|
||||
࠸
|
||||
࠹
|
||||
࠺
|
||||
࠻
|
||||
࠼
|
||||
࠽
|
||||
࠾
|
||||
68
ppocr/utils/dict/spin_dict.txt
Normal file
68
ppocr/utils/dict/spin_dict.txt
Normal file
@@ -0,0 +1,68 @@
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
:
|
||||
(
|
||||
'
|
||||
-
|
||||
,
|
||||
%
|
||||
>
|
||||
.
|
||||
[
|
||||
?
|
||||
)
|
||||
"
|
||||
=
|
||||
_
|
||||
*
|
||||
]
|
||||
;
|
||||
&
|
||||
+
|
||||
$
|
||||
@
|
||||
/
|
||||
|
|
||||
!
|
||||
<
|
||||
#
|
||||
`
|
||||
{
|
||||
~
|
||||
\
|
||||
}
|
||||
^
|
||||
157
ppocr/utils/dict/syriac_dict.txt
Normal file
157
ppocr/utils/dict/syriac_dict.txt
Normal file
@@ -0,0 +1,157 @@
|
||||
!
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
:
|
||||
?
|
||||
@
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
_
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
É
|
||||
é
|
||||
܀
|
||||
܁
|
||||
܂
|
||||
܃
|
||||
܄
|
||||
܅
|
||||
܆
|
||||
܇
|
||||
܈
|
||||
܉
|
||||
܊
|
||||
܋
|
||||
܌
|
||||
܍
|
||||
|
||||
ܐ
|
||||
ܑ
|
||||
ܒ
|
||||
ܓ
|
||||
ܔ
|
||||
ܕ
|
||||
ܖ
|
||||
ܗ
|
||||
ܘ
|
||||
ܙ
|
||||
ܚ
|
||||
ܛ
|
||||
ܜ
|
||||
ܝ
|
||||
ܞ
|
||||
ܟ
|
||||
ܠ
|
||||
ܡ
|
||||
ܢ
|
||||
ܣ
|
||||
ܤ
|
||||
ܥ
|
||||
ܦ
|
||||
ܧ
|
||||
ܨ
|
||||
ܩ
|
||||
ܪ
|
||||
ܫ
|
||||
ܬ
|
||||
ܭ
|
||||
ܮ
|
||||
ܯ
|
||||
ܰ
|
||||
ܱ
|
||||
ܲ
|
||||
ܳ
|
||||
ܴ
|
||||
ܵ
|
||||
ܶ
|
||||
ܷ
|
||||
ܸ
|
||||
ܹ
|
||||
ܺ
|
||||
ܻ
|
||||
ܼ
|
||||
ܽ
|
||||
ܾ
|
||||
ܿ
|
||||
݀
|
||||
݁
|
||||
݂
|
||||
݃
|
||||
݄
|
||||
݅
|
||||
݆
|
||||
݇
|
||||
݈
|
||||
݉
|
||||
݊
|
||||
ݍ
|
||||
ݎ
|
||||
ݏ
|
||||
128
ppocr/utils/dict/ta_dict.txt
Normal file
128
ppocr/utils/dict/ta_dict.txt
Normal file
@@ -0,0 +1,128 @@
|
||||
t
|
||||
a
|
||||
_
|
||||
i
|
||||
m
|
||||
g
|
||||
/
|
||||
3
|
||||
I
|
||||
L
|
||||
S
|
||||
V
|
||||
R
|
||||
C
|
||||
2
|
||||
0
|
||||
1
|
||||
v
|
||||
l
|
||||
9
|
||||
7
|
||||
8
|
||||
.
|
||||
j
|
||||
p
|
||||
ப
|
||||
ூ
|
||||
த
|
||||
ம
|
||||
ி
|
||||
வ
|
||||
ர
|
||||
்
|
||||
ந
|
||||
ோ
|
||||
ன
|
||||
6
|
||||
ஆ
|
||||
ற
|
||||
ல
|
||||
5
|
||||
ள
|
||||
ா
|
||||
ொ
|
||||
ழ
|
||||
ு
|
||||
4
|
||||
ெ
|
||||
ண
|
||||
க
|
||||
ட
|
||||
ை
|
||||
ே
|
||||
ச
|
||||
ய
|
||||
ஒ
|
||||
இ
|
||||
அ
|
||||
ங
|
||||
உ
|
||||
ீ
|
||||
ஞ
|
||||
எ
|
||||
ஓ
|
||||
ஃ
|
||||
ஜ
|
||||
ஷ
|
||||
ஸ
|
||||
ஏ
|
||||
ஊ
|
||||
ஹ
|
||||
ஈ
|
||||
ஐ
|
||||
ௌ
|
||||
ஔ
|
||||
s
|
||||
c
|
||||
e
|
||||
n
|
||||
w
|
||||
F
|
||||
T
|
||||
O
|
||||
P
|
||||
K
|
||||
A
|
||||
N
|
||||
G
|
||||
Y
|
||||
E
|
||||
M
|
||||
H
|
||||
U
|
||||
B
|
||||
o
|
||||
b
|
||||
D
|
||||
d
|
||||
r
|
||||
W
|
||||
u
|
||||
y
|
||||
f
|
||||
X
|
||||
k
|
||||
q
|
||||
h
|
||||
J
|
||||
z
|
||||
Z
|
||||
Q
|
||||
x
|
||||
-
|
||||
'
|
||||
$
|
||||
,
|
||||
%
|
||||
@
|
||||
é
|
||||
!
|
||||
#
|
||||
+
|
||||
É
|
||||
&
|
||||
:
|
||||
(
|
||||
?
|
||||
|
||||
277
ppocr/utils/dict/table_dict.txt
Normal file
277
ppocr/utils/dict/table_dict.txt
Normal file
@@ -0,0 +1,277 @@
|
||||
←
|
||||
</overline>
|
||||
☆
|
||||
─
|
||||
α
|
||||
|
||||
|
||||
⋅
|
||||
$
|
||||
ω
|
||||
ψ
|
||||
χ
|
||||
(
|
||||
υ
|
||||
≥
|
||||
σ
|
||||
,
|
||||
ρ
|
||||
ε
|
||||
0
|
||||
■
|
||||
4
|
||||
8
|
||||
✗
|
||||
b
|
||||
<
|
||||
✓
|
||||
Ψ
|
||||
Ω
|
||||
€
|
||||
D
|
||||
3
|
||||
Π
|
||||
H
|
||||
║
|
||||
</strike>
|
||||
L
|
||||
Φ
|
||||
Χ
|
||||
θ
|
||||
P
|
||||
κ
|
||||
λ
|
||||
μ
|
||||
T
|
||||
ξ
|
||||
X
|
||||
β
|
||||
γ
|
||||
δ
|
||||
\
|
||||
ζ
|
||||
η
|
||||
`
|
||||
d
|
||||
<strike>
|
||||
h
|
||||
f
|
||||
l
|
||||
Θ
|
||||
p
|
||||
√
|
||||
t
|
||||
</sub>
|
||||
x
|
||||
Β
|
||||
Γ
|
||||
Δ
|
||||
|
|
||||
ǂ
|
||||
ɛ
|
||||
j
|
||||
̧
|
||||
➢
|
||||
|
||||
̌
|
||||
′
|
||||
«
|
||||
△
|
||||
▲
|
||||
#
|
||||
</b>
|
||||
'
|
||||
Ι
|
||||
+
|
||||
¶
|
||||
/
|
||||
▼
|
||||
⇑
|
||||
□
|
||||
·
|
||||
7
|
||||
▪
|
||||
;
|
||||
?
|
||||
➔
|
||||
∩
|
||||
C
|
||||
÷
|
||||
G
|
||||
⇒
|
||||
K
|
||||
<sup>
|
||||
O
|
||||
S
|
||||
С
|
||||
W
|
||||
Α
|
||||
[
|
||||
○
|
||||
_
|
||||
●
|
||||
‡
|
||||
c
|
||||
z
|
||||
g
|
||||
<i>
|
||||
o
|
||||
<sub>
|
||||
〈
|
||||
〉
|
||||
s
|
||||
⩽
|
||||
w
|
||||
φ
|
||||
ʹ
|
||||
{
|
||||
»
|
||||
∣
|
||||
̆
|
||||
e
|
||||
ˆ
|
||||
∈
|
||||
τ
|
||||
◆
|
||||
ι
|
||||
∅
|
||||
∆
|
||||
∙
|
||||
∘
|
||||
Ø
|
||||
ß
|
||||
✔
|
||||
∞
|
||||
∑
|
||||
−
|
||||
×
|
||||
◊
|
||||
∗
|
||||
∖
|
||||
˃
|
||||
˂
|
||||
∫
|
||||
"
|
||||
i
|
||||
&
|
||||
π
|
||||
↔
|
||||
*
|
||||
∥
|
||||
æ
|
||||
∧
|
||||
.
|
||||
⁄
|
||||
ø
|
||||
Q
|
||||
∼
|
||||
6
|
||||
⁎
|
||||
:
|
||||
★
|
||||
>
|
||||
a
|
||||
B
|
||||
≈
|
||||
F
|
||||
J
|
||||
̄
|
||||
N
|
||||
♯
|
||||
R
|
||||
V
|
||||
<overline>
|
||||
―
|
||||
Z
|
||||
♣
|
||||
^
|
||||
¤
|
||||
¥
|
||||
§
|
||||
<underline>
|
||||
¢
|
||||
£
|
||||
≦
|
||||
|
||||
≤
|
||||
‖
|
||||
Λ
|
||||
©
|
||||
n
|
||||
↓
|
||||
→
|
||||
↑
|
||||
r
|
||||
°
|
||||
±
|
||||
v
|
||||
<b>
|
||||
♂
|
||||
k
|
||||
♀
|
||||
~
|
||||
ᅟ
|
||||
̇
|
||||
@
|
||||
”
|
||||
♦
|
||||
ł
|
||||
®
|
||||
⊕
|
||||
„
|
||||
!
|
||||
</sup>
|
||||
%
|
||||
⇓
|
||||
)
|
||||
-
|
||||
1
|
||||
5
|
||||
9
|
||||
=
|
||||
А
|
||||
A
|
||||
‰
|
||||
⋆
|
||||
Σ
|
||||
E
|
||||
◦
|
||||
I
|
||||
※
|
||||
M
|
||||
m
|
||||
̨
|
||||
⩾
|
||||
†
|
||||
</i>
|
||||
•
|
||||
U
|
||||
Y
|
||||
|
||||
]
|
||||
̸
|
||||
2
|
||||
‐
|
||||
–
|
||||
‒
|
||||
̂
|
||||
—
|
||||
̀
|
||||
́
|
||||
’
|
||||
‘
|
||||
⋮
|
||||
⋯
|
||||
̊
|
||||
“
|
||||
̈
|
||||
≧
|
||||
q
|
||||
u
|
||||
ı
|
||||
y
|
||||
</underline>
|
||||
|
||||
̃
|
||||
}
|
||||
ν
|
||||
39
ppocr/utils/dict/table_master_structure_dict.txt
Normal file
39
ppocr/utils/dict/table_master_structure_dict.txt
Normal file
@@ -0,0 +1,39 @@
|
||||
<thead>
|
||||
<tr>
|
||||
<td></td>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
<eb></eb>
|
||||
</tbody>
|
||||
<td
|
||||
colspan="5"
|
||||
>
|
||||
</td>
|
||||
colspan="2"
|
||||
colspan="3"
|
||||
<eb2></eb2>
|
||||
<eb1></eb1>
|
||||
rowspan="2"
|
||||
colspan="4"
|
||||
colspan="6"
|
||||
rowspan="3"
|
||||
colspan="9"
|
||||
colspan="10"
|
||||
colspan="7"
|
||||
rowspan="4"
|
||||
rowspan="5"
|
||||
rowspan="9"
|
||||
colspan="8"
|
||||
rowspan="8"
|
||||
rowspan="6"
|
||||
rowspan="7"
|
||||
rowspan="10"
|
||||
<eb3></eb3>
|
||||
<eb4></eb4>
|
||||
<eb5></eb5>
|
||||
<eb6></eb6>
|
||||
<eb7></eb7>
|
||||
<eb8></eb8>
|
||||
<eb9></eb9>
|
||||
<eb10></eb10>
|
||||
28
ppocr/utils/dict/table_structure_dict.txt
Normal file
28
ppocr/utils/dict/table_structure_dict.txt
Normal file
@@ -0,0 +1,28 @@
|
||||
<thead>
|
||||
<tr>
|
||||
<td>
|
||||
</td>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
</tbody>
|
||||
<td
|
||||
colspan="5"
|
||||
>
|
||||
colspan="2"
|
||||
colspan="3"
|
||||
rowspan="2"
|
||||
colspan="4"
|
||||
colspan="6"
|
||||
rowspan="3"
|
||||
colspan="9"
|
||||
colspan="10"
|
||||
colspan="7"
|
||||
rowspan="4"
|
||||
rowspan="5"
|
||||
rowspan="9"
|
||||
colspan="8"
|
||||
rowspan="8"
|
||||
rowspan="6"
|
||||
rowspan="7"
|
||||
rowspan="10"
|
||||
48
ppocr/utils/dict/table_structure_dict_ch.txt
Normal file
48
ppocr/utils/dict/table_structure_dict_ch.txt
Normal file
@@ -0,0 +1,48 @@
|
||||
<thead>
|
||||
</thead>
|
||||
<tbody>
|
||||
</tbody>
|
||||
<tr>
|
||||
</tr>
|
||||
<td>
|
||||
<td
|
||||
>
|
||||
</td>
|
||||
colspan="2"
|
||||
colspan="3"
|
||||
colspan="4"
|
||||
colspan="5"
|
||||
colspan="6"
|
||||
colspan="7"
|
||||
colspan="8"
|
||||
colspan="9"
|
||||
colspan="10"
|
||||
colspan="11"
|
||||
colspan="12"
|
||||
colspan="13"
|
||||
colspan="14"
|
||||
colspan="15"
|
||||
colspan="16"
|
||||
colspan="17"
|
||||
colspan="18"
|
||||
colspan="19"
|
||||
colspan="20"
|
||||
rowspan="2"
|
||||
rowspan="3"
|
||||
rowspan="4"
|
||||
rowspan="5"
|
||||
rowspan="6"
|
||||
rowspan="7"
|
||||
rowspan="8"
|
||||
rowspan="9"
|
||||
rowspan="10"
|
||||
rowspan="11"
|
||||
rowspan="12"
|
||||
rowspan="13"
|
||||
rowspan="14"
|
||||
rowspan="15"
|
||||
rowspan="16"
|
||||
rowspan="17"
|
||||
rowspan="18"
|
||||
rowspan="19"
|
||||
rowspan="20"
|
||||
151
ppocr/utils/dict/te_dict.txt
Normal file
151
ppocr/utils/dict/te_dict.txt
Normal file
@@ -0,0 +1,151 @@
|
||||
t
|
||||
e
|
||||
_
|
||||
i
|
||||
m
|
||||
g
|
||||
/
|
||||
5
|
||||
I
|
||||
L
|
||||
S
|
||||
V
|
||||
R
|
||||
C
|
||||
2
|
||||
0
|
||||
1
|
||||
v
|
||||
a
|
||||
l
|
||||
3
|
||||
4
|
||||
8
|
||||
9
|
||||
.
|
||||
j
|
||||
p
|
||||
త
|
||||
ె
|
||||
ర
|
||||
క
|
||||
్
|
||||
ి
|
||||
ం
|
||||
చ
|
||||
ే
|
||||
ద
|
||||
ు
|
||||
7
|
||||
6
|
||||
ఉ
|
||||
ా
|
||||
మ
|
||||
ట
|
||||
ో
|
||||
వ
|
||||
ప
|
||||
ల
|
||||
శ
|
||||
ఆ
|
||||
య
|
||||
ై
|
||||
భ
|
||||
'
|
||||
ీ
|
||||
గ
|
||||
ూ
|
||||
డ
|
||||
ధ
|
||||
హ
|
||||
న
|
||||
జ
|
||||
స
|
||||
[
|
||||
|
||||
ష
|
||||
అ
|
||||
ణ
|
||||
ఫ
|
||||
బ
|
||||
ఎ
|
||||
;
|
||||
ళ
|
||||
థ
|
||||
ొ
|
||||
ఠ
|
||||
ృ
|
||||
ఒ
|
||||
ఇ
|
||||
ః
|
||||
ఊ
|
||||
ఖ
|
||||
-
|
||||
ఐ
|
||||
ఘ
|
||||
ౌ
|
||||
ఏ
|
||||
ఈ
|
||||
ఛ
|
||||
,
|
||||
ఓ
|
||||
ఞ
|
||||
|
|
||||
?
|
||||
:
|
||||
ఢ
|
||||
"
|
||||
(
|
||||
”
|
||||
!
|
||||
+
|
||||
)
|
||||
*
|
||||
=
|
||||
&
|
||||
“
|
||||
€
|
||||
]
|
||||
£
|
||||
$
|
||||
s
|
||||
c
|
||||
n
|
||||
w
|
||||
k
|
||||
J
|
||||
G
|
||||
u
|
||||
d
|
||||
r
|
||||
E
|
||||
o
|
||||
h
|
||||
y
|
||||
b
|
||||
f
|
||||
B
|
||||
M
|
||||
O
|
||||
T
|
||||
N
|
||||
D
|
||||
P
|
||||
A
|
||||
F
|
||||
x
|
||||
W
|
||||
Y
|
||||
U
|
||||
H
|
||||
K
|
||||
X
|
||||
z
|
||||
Z
|
||||
Q
|
||||
q
|
||||
É
|
||||
%
|
||||
#
|
||||
@
|
||||
é
|
||||
81
ppocr/utils/dict/th_dict.txt
Normal file
81
ppocr/utils/dict/th_dict.txt
Normal file
@@ -0,0 +1,81 @@
|
||||
ก
|
||||
ข
|
||||
ฃ
|
||||
ค
|
||||
ฅ
|
||||
ฆ
|
||||
ง
|
||||
จ
|
||||
ฉ
|
||||
ช
|
||||
ซ
|
||||
ฌ
|
||||
ญ
|
||||
ฎ
|
||||
ฏ
|
||||
ฐ
|
||||
ฑ
|
||||
ฒ
|
||||
ณ
|
||||
ด
|
||||
ต
|
||||
ถ
|
||||
ท
|
||||
ธ
|
||||
น
|
||||
บ
|
||||
ป
|
||||
ผ
|
||||
ฝ
|
||||
พ
|
||||
ฟ
|
||||
ภ
|
||||
ม
|
||||
ย
|
||||
ร
|
||||
ล
|
||||
ว
|
||||
ศ
|
||||
ษ
|
||||
ส
|
||||
ห
|
||||
ฬ
|
||||
อ
|
||||
ฮ
|
||||
ะ
|
||||
ั
|
||||
า
|
||||
ำ
|
||||
ิ
|
||||
ี
|
||||
ึ
|
||||
ื
|
||||
ุ
|
||||
ู
|
||||
เ
|
||||
แ
|
||||
โ
|
||||
ใ
|
||||
ไ
|
||||
็
|
||||
่
|
||||
้
|
||||
๊
|
||||
๋
|
||||
์
|
||||
๐
|
||||
๑
|
||||
๒
|
||||
๓
|
||||
๔
|
||||
๕
|
||||
๖
|
||||
๗
|
||||
๘
|
||||
๙
|
||||
ฯ
|
||||
ๆ
|
||||
ฤ
|
||||
ฤา
|
||||
ฦ
|
||||
ฦา
|
||||
131
ppocr/utils/dict/ug_dict.txt
Normal file
131
ppocr/utils/dict/ug_dict.txt
Normal file
@@ -0,0 +1,131 @@
|
||||
`
|
||||
~
|
||||
=
|
||||
|
|
||||
|
||||
_
|
||||
-
|
||||
,
|
||||
;
|
||||
:
|
||||
!
|
||||
?
|
||||
/
|
||||
.
|
||||
'
|
||||
"
|
||||
(
|
||||
)
|
||||
[
|
||||
]
|
||||
{
|
||||
}
|
||||
@
|
||||
$
|
||||
*
|
||||
\
|
||||
&
|
||||
#
|
||||
%
|
||||
+
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
a
|
||||
A
|
||||
b
|
||||
B
|
||||
c
|
||||
C
|
||||
d
|
||||
D
|
||||
e
|
||||
E
|
||||
f
|
||||
F
|
||||
g
|
||||
G
|
||||
h
|
||||
H
|
||||
i
|
||||
I
|
||||
j
|
||||
J
|
||||
k
|
||||
K
|
||||
l
|
||||
L
|
||||
m
|
||||
M
|
||||
n
|
||||
N
|
||||
o
|
||||
O
|
||||
p
|
||||
P
|
||||
q
|
||||
Q
|
||||
r
|
||||
R
|
||||
s
|
||||
S
|
||||
t
|
||||
T
|
||||
u
|
||||
U
|
||||
v
|
||||
V
|
||||
w
|
||||
W
|
||||
x
|
||||
X
|
||||
y
|
||||
Y
|
||||
z
|
||||
Z
|
||||
Ö
|
||||
ö
|
||||
Ü
|
||||
ü
|
||||
Ë
|
||||
ë
|
||||
ا
|
||||
ە
|
||||
ب
|
||||
پ
|
||||
ت
|
||||
ج
|
||||
چ
|
||||
خ
|
||||
د
|
||||
ر
|
||||
ز
|
||||
ژ
|
||||
س
|
||||
ش
|
||||
غ
|
||||
ف
|
||||
ق
|
||||
ك
|
||||
گ
|
||||
ڭ
|
||||
ل
|
||||
م
|
||||
ن
|
||||
ھ
|
||||
و
|
||||
ۇ
|
||||
ۆ
|
||||
ۈ
|
||||
ۋ
|
||||
ې
|
||||
ى
|
||||
ي
|
||||
ئ
|
||||
142
ppocr/utils/dict/uk_dict.txt
Normal file
142
ppocr/utils/dict/uk_dict.txt
Normal file
@@ -0,0 +1,142 @@
|
||||
u
|
||||
k
|
||||
_
|
||||
i
|
||||
m
|
||||
g
|
||||
/
|
||||
1
|
||||
6
|
||||
I
|
||||
L
|
||||
S
|
||||
V
|
||||
R
|
||||
C
|
||||
2
|
||||
0
|
||||
v
|
||||
a
|
||||
l
|
||||
7
|
||||
9
|
||||
.
|
||||
j
|
||||
p
|
||||
в
|
||||
і
|
||||
д
|
||||
п
|
||||
о
|
||||
н
|
||||
с
|
||||
т
|
||||
ю
|
||||
4
|
||||
5
|
||||
3
|
||||
а
|
||||
и
|
||||
м
|
||||
е
|
||||
р
|
||||
ч
|
||||
у
|
||||
Б
|
||||
з
|
||||
л
|
||||
к
|
||||
8
|
||||
А
|
||||
В
|
||||
г
|
||||
є
|
||||
б
|
||||
ь
|
||||
х
|
||||
ґ
|
||||
ш
|
||||
ц
|
||||
ф
|
||||
я
|
||||
щ
|
||||
ж
|
||||
Г
|
||||
Х
|
||||
У
|
||||
Т
|
||||
Е
|
||||
І
|
||||
Н
|
||||
П
|
||||
З
|
||||
Л
|
||||
Ю
|
||||
С
|
||||
Д
|
||||
М
|
||||
К
|
||||
Р
|
||||
Ф
|
||||
О
|
||||
Ц
|
||||
И
|
||||
Я
|
||||
Ч
|
||||
Ш
|
||||
Ж
|
||||
Є
|
||||
Ґ
|
||||
Ь
|
||||
s
|
||||
c
|
||||
e
|
||||
n
|
||||
w
|
||||
A
|
||||
P
|
||||
r
|
||||
E
|
||||
t
|
||||
o
|
||||
h
|
||||
d
|
||||
y
|
||||
M
|
||||
G
|
||||
N
|
||||
F
|
||||
B
|
||||
T
|
||||
D
|
||||
U
|
||||
O
|
||||
W
|
||||
Z
|
||||
f
|
||||
H
|
||||
Y
|
||||
b
|
||||
K
|
||||
z
|
||||
x
|
||||
Q
|
||||
X
|
||||
q
|
||||
J
|
||||
$
|
||||
-
|
||||
'
|
||||
#
|
||||
&
|
||||
%
|
||||
?
|
||||
:
|
||||
!
|
||||
,
|
||||
+
|
||||
@
|
||||
(
|
||||
é
|
||||
É
|
||||
|
||||
100067
ppocr/utils/dict/unimernet_tokenizer/tokenizer.json
Normal file
100067
ppocr/utils/dict/unimernet_tokenizer/tokenizer.json
Normal file
File diff suppressed because it is too large
Load Diff
205
ppocr/utils/dict/unimernet_tokenizer/tokenizer_config.json
Normal file
205
ppocr/utils/dict/unimernet_tokenizer/tokenizer_config.json
Normal file
@@ -0,0 +1,205 @@
|
||||
{
|
||||
"added_tokens_decoder": {
|
||||
"0": {
|
||||
"content": "<s>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"1": {
|
||||
"content": "<pad>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"2": {
|
||||
"content": "</s>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"3": {
|
||||
"content": "<unk>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"4": {
|
||||
"content": "[START_REF]",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"5": {
|
||||
"content": "[END_REF]",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"6": {
|
||||
"content": "[IMAGE]",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"7": {
|
||||
"content": "<fragments>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"8": {
|
||||
"content": "</fragments>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"9": {
|
||||
"content": "<work>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"10": {
|
||||
"content": "</work>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"11": {
|
||||
"content": "[START_SUP]",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"12": {
|
||||
"content": "[END_SUP]",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"13": {
|
||||
"content": "[START_SUB]",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"14": {
|
||||
"content": "[END_SUB]",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"15": {
|
||||
"content": "[START_DNA]",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"16": {
|
||||
"content": "[END_DNA]",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"17": {
|
||||
"content": "[START_AMINO]",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"18": {
|
||||
"content": "[END_AMINO]",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"19": {
|
||||
"content": "[START_SMILES]",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"20": {
|
||||
"content": "[END_SMILES]",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"21": {
|
||||
"content": "[START_I_SMILES]",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"22": {
|
||||
"content": "[END_I_SMILES]",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
}
|
||||
},
|
||||
"additional_special_tokens": [],
|
||||
"bos_token": "<s>",
|
||||
"clean_up_tokenization_spaces": false,
|
||||
"eos_token": "</s>",
|
||||
"max_length": 4096,
|
||||
"model_max_length": 768,
|
||||
"pad_to_multiple_of": null,
|
||||
"pad_token": "<pad>",
|
||||
"pad_token_type_id": 0,
|
||||
"padding_side": "right",
|
||||
"processor_class": "VariableDonutProcessor",
|
||||
"stride": 0,
|
||||
"tokenizer_class": "NougatTokenizer",
|
||||
"truncation_side": "right",
|
||||
"truncation_strategy": "longest_first",
|
||||
"unk_token": "<unk>",
|
||||
"vocab_file": null
|
||||
}
|
||||
137
ppocr/utils/dict/ur_dict.txt
Normal file
137
ppocr/utils/dict/ur_dict.txt
Normal file
@@ -0,0 +1,137 @@
|
||||
u
|
||||
r
|
||||
_
|
||||
i
|
||||
m
|
||||
g
|
||||
/
|
||||
3
|
||||
I
|
||||
L
|
||||
S
|
||||
V
|
||||
R
|
||||
C
|
||||
2
|
||||
0
|
||||
1
|
||||
v
|
||||
a
|
||||
l
|
||||
9
|
||||
7
|
||||
8
|
||||
.
|
||||
j
|
||||
p
|
||||
|
||||
چ
|
||||
ٹ
|
||||
پ
|
||||
ا
|
||||
ئ
|
||||
ی
|
||||
ے
|
||||
4
|
||||
6
|
||||
و
|
||||
ل
|
||||
ن
|
||||
ڈ
|
||||
ھ
|
||||
ک
|
||||
ت
|
||||
ش
|
||||
ف
|
||||
ق
|
||||
ر
|
||||
د
|
||||
5
|
||||
ب
|
||||
ج
|
||||
خ
|
||||
ہ
|
||||
س
|
||||
ز
|
||||
غ
|
||||
ڑ
|
||||
ں
|
||||
آ
|
||||
م
|
||||
ؤ
|
||||
ط
|
||||
ص
|
||||
ح
|
||||
ع
|
||||
گ
|
||||
ث
|
||||
ض
|
||||
ذ
|
||||
ۓ
|
||||
ِ
|
||||
ء
|
||||
ظ
|
||||
ً
|
||||
ي
|
||||
ُ
|
||||
ۃ
|
||||
أ
|
||||
ٰ
|
||||
ە
|
||||
ژ
|
||||
ۂ
|
||||
ة
|
||||
ّ
|
||||
ك
|
||||
ه
|
||||
s
|
||||
c
|
||||
e
|
||||
n
|
||||
w
|
||||
o
|
||||
d
|
||||
t
|
||||
D
|
||||
M
|
||||
T
|
||||
U
|
||||
E
|
||||
b
|
||||
P
|
||||
h
|
||||
y
|
||||
W
|
||||
H
|
||||
A
|
||||
x
|
||||
B
|
||||
O
|
||||
N
|
||||
G
|
||||
Y
|
||||
Q
|
||||
F
|
||||
k
|
||||
K
|
||||
q
|
||||
J
|
||||
Z
|
||||
f
|
||||
z
|
||||
X
|
||||
'
|
||||
@
|
||||
&
|
||||
!
|
||||
,
|
||||
:
|
||||
$
|
||||
-
|
||||
#
|
||||
?
|
||||
%
|
||||
é
|
||||
+
|
||||
(
|
||||
É
|
||||
113
ppocr/utils/dict/vi_dict.txt
Normal file
113
ppocr/utils/dict/vi_dict.txt
Normal file
@@ -0,0 +1,113 @@
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
g
|
||||
h
|
||||
i
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
à
|
||||
á
|
||||
â
|
||||
ã
|
||||
è
|
||||
é
|
||||
ê
|
||||
ì
|
||||
í
|
||||
ò
|
||||
ó
|
||||
ô
|
||||
õ
|
||||
ù
|
||||
ú
|
||||
ý
|
||||
ă
|
||||
Đ
|
||||
đ
|
||||
ĩ
|
||||
ũ
|
||||
ơ
|
||||
ư
|
||||
ạ
|
||||
ả
|
||||
ấ
|
||||
ầ
|
||||
ẩ
|
||||
ẫ
|
||||
ậ
|
||||
ắ
|
||||
ằ
|
||||
ẳ
|
||||
ẵ
|
||||
ặ
|
||||
ẹ
|
||||
ẻ
|
||||
ẽ
|
||||
ế
|
||||
ề
|
||||
ể
|
||||
ễ
|
||||
ệ
|
||||
ỉ
|
||||
ị
|
||||
ọ
|
||||
ỏ
|
||||
ố
|
||||
ồ
|
||||
ổ
|
||||
ỗ
|
||||
ộ
|
||||
ớ
|
||||
ờ
|
||||
ở
|
||||
ỡ
|
||||
ợ
|
||||
ụ
|
||||
ủ
|
||||
ứ
|
||||
ừ
|
||||
ử
|
||||
ữ
|
||||
ự
|
||||
ỳ
|
||||
ỵ
|
||||
ỷ
|
||||
ỹ
|
||||
110
ppocr/utils/dict/xi_dict.txt
Normal file
110
ppocr/utils/dict/xi_dict.txt
Normal file
@@ -0,0 +1,110 @@
|
||||
x
|
||||
i
|
||||
_
|
||||
m
|
||||
g
|
||||
/
|
||||
1
|
||||
0
|
||||
I
|
||||
L
|
||||
S
|
||||
V
|
||||
R
|
||||
C
|
||||
2
|
||||
v
|
||||
a
|
||||
l
|
||||
3
|
||||
6
|
||||
4
|
||||
5
|
||||
.
|
||||
j
|
||||
p
|
||||
|
||||
Q
|
||||
u
|
||||
e
|
||||
r
|
||||
o
|
||||
8
|
||||
7
|
||||
n
|
||||
c
|
||||
9
|
||||
t
|
||||
b
|
||||
é
|
||||
q
|
||||
d
|
||||
ó
|
||||
y
|
||||
F
|
||||
s
|
||||
,
|
||||
O
|
||||
í
|
||||
T
|
||||
f
|
||||
"
|
||||
U
|
||||
M
|
||||
h
|
||||
:
|
||||
P
|
||||
H
|
||||
A
|
||||
E
|
||||
D
|
||||
z
|
||||
N
|
||||
á
|
||||
ñ
|
||||
ú
|
||||
%
|
||||
;
|
||||
è
|
||||
+
|
||||
Y
|
||||
-
|
||||
B
|
||||
G
|
||||
(
|
||||
)
|
||||
¿
|
||||
?
|
||||
w
|
||||
¡
|
||||
!
|
||||
X
|
||||
É
|
||||
K
|
||||
k
|
||||
Á
|
||||
ü
|
||||
Ú
|
||||
«
|
||||
»
|
||||
J
|
||||
'
|
||||
ö
|
||||
W
|
||||
Z
|
||||
º
|
||||
Ö
|
||||
|
||||
[
|
||||
]
|
||||
Ç
|
||||
ç
|
||||
à
|
||||
ä
|
||||
û
|
||||
ò
|
||||
Í
|
||||
ê
|
||||
ô
|
||||
ø
|
||||
ª
|
||||
90
ppocr/utils/dict90.txt
Normal file
90
ppocr/utils/dict90.txt
Normal file
@@ -0,0 +1,90 @@
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
!
|
||||
"
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
)
|
||||
*
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
:
|
||||
;
|
||||
<
|
||||
=
|
||||
>
|
||||
?
|
||||
@
|
||||
[
|
||||
\
|
||||
]
|
||||
_
|
||||
`
|
||||
~
|
||||
852
ppocr/utils/e2e_metric/Deteval.py
Executable file
852
ppocr/utils/e2e_metric/Deteval.py
Executable file
@@ -0,0 +1,852 @@
|
||||
# Copyright (c) 2021 PaddlePaddle Authors. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import json
|
||||
import numpy as np
|
||||
import scipy.io as io
|
||||
|
||||
from ppocr.utils.utility import check_install
|
||||
|
||||
from ppocr.utils.e2e_metric.polygon_fast import iod, area_of_intersection, area
|
||||
|
||||
|
||||
def get_socre_A(gt_dir, pred_dict):
|
||||
allInputs = 1
|
||||
|
||||
def input_reading_mod(pred_dict):
|
||||
"""This helper reads input from txt files"""
|
||||
det = []
|
||||
n = len(pred_dict)
|
||||
for i in range(n):
|
||||
points = pred_dict[i]["points"]
|
||||
text = pred_dict[i]["texts"]
|
||||
point = ",".join(
|
||||
map(
|
||||
str,
|
||||
points.reshape(
|
||||
-1,
|
||||
),
|
||||
)
|
||||
)
|
||||
det.append([point, text])
|
||||
return det
|
||||
|
||||
def gt_reading_mod(gt_dict):
|
||||
"""This helper reads groundtruths from mat files"""
|
||||
gt = []
|
||||
n = len(gt_dict)
|
||||
for i in range(n):
|
||||
points = gt_dict[i]["points"].tolist()
|
||||
h = len(points)
|
||||
text = gt_dict[i]["text"]
|
||||
xx = [
|
||||
np.array(["x:"], dtype="<U2"),
|
||||
0,
|
||||
np.array(["y:"], dtype="<U2"),
|
||||
0,
|
||||
np.array(["#"], dtype="<U1"),
|
||||
np.array(["#"], dtype="<U1"),
|
||||
]
|
||||
t_x, t_y = [], []
|
||||
for j in range(h):
|
||||
t_x.append(points[j][0])
|
||||
t_y.append(points[j][1])
|
||||
xx[1] = np.array([t_x], dtype="int16")
|
||||
xx[3] = np.array([t_y], dtype="int16")
|
||||
if text != "":
|
||||
xx[4] = np.array([text], dtype="U{}".format(len(text)))
|
||||
xx[5] = np.array(["c"], dtype="<U1")
|
||||
gt.append(xx)
|
||||
return gt
|
||||
|
||||
def detection_filtering(detections, groundtruths, threshold=0.5):
|
||||
for gt_id, gt in enumerate(groundtruths):
|
||||
if (gt[5] == "#") and (gt[1].shape[1] > 1):
|
||||
gt_x = list(map(int, np.squeeze(gt[1])))
|
||||
gt_y = list(map(int, np.squeeze(gt[3])))
|
||||
for det_id, detection in enumerate(detections):
|
||||
detection_orig = detection
|
||||
detection = [float(x) for x in detection[0].split(",")]
|
||||
detection = list(map(int, detection))
|
||||
det_x = detection[0::2]
|
||||
det_y = detection[1::2]
|
||||
det_gt_iou = iod(det_x, det_y, gt_x, gt_y)
|
||||
if det_gt_iou > threshold:
|
||||
detections[det_id] = []
|
||||
|
||||
detections[:] = [item for item in detections if item != []]
|
||||
return detections
|
||||
|
||||
def sigma_calculation(det_x, det_y, gt_x, gt_y):
|
||||
"""
|
||||
sigma = inter_area / gt_area
|
||||
"""
|
||||
return np.round(
|
||||
(area_of_intersection(det_x, det_y, gt_x, gt_y) / area(gt_x, gt_y)), 2
|
||||
)
|
||||
|
||||
def tau_calculation(det_x, det_y, gt_x, gt_y):
|
||||
if area(det_x, det_y) == 0.0:
|
||||
return 0
|
||||
return np.round(
|
||||
(area_of_intersection(det_x, det_y, gt_x, gt_y) / area(det_x, det_y)), 2
|
||||
)
|
||||
|
||||
##############################Initialization###################################
|
||||
# global_sigma = []
|
||||
# global_tau = []
|
||||
# global_pred_str = []
|
||||
# global_gt_str = []
|
||||
###############################################################################
|
||||
|
||||
for input_id in range(allInputs):
|
||||
if (
|
||||
(input_id != ".DS_Store")
|
||||
and (input_id != "Pascal_result.txt")
|
||||
and (input_id != "Pascal_result_curved.txt")
|
||||
and (input_id != "Pascal_result_non_curved.txt")
|
||||
and (input_id != "Deteval_result.txt")
|
||||
and (input_id != "Deteval_result_curved.txt")
|
||||
and (input_id != "Deteval_result_non_curved.txt")
|
||||
):
|
||||
detections = input_reading_mod(pred_dict)
|
||||
groundtruths = gt_reading_mod(gt_dir)
|
||||
detections = detection_filtering(
|
||||
detections, groundtruths
|
||||
) # filters detections overlapping with DC area
|
||||
dc_id = []
|
||||
for i in range(len(groundtruths)):
|
||||
if groundtruths[i][5] == "#":
|
||||
dc_id.append(i)
|
||||
cnt = 0
|
||||
for a in dc_id:
|
||||
num = a - cnt
|
||||
del groundtruths[num]
|
||||
cnt += 1
|
||||
|
||||
local_sigma_table = np.zeros((len(groundtruths), len(detections)))
|
||||
local_tau_table = np.zeros((len(groundtruths), len(detections)))
|
||||
local_pred_str = {}
|
||||
local_gt_str = {}
|
||||
|
||||
for gt_id, gt in enumerate(groundtruths):
|
||||
if len(detections) > 0:
|
||||
for det_id, detection in enumerate(detections):
|
||||
detection_orig = detection
|
||||
detection = [float(x) for x in detection[0].split(",")]
|
||||
detection = list(map(int, detection))
|
||||
pred_seq_str = detection_orig[1].strip()
|
||||
det_x = detection[0::2]
|
||||
det_y = detection[1::2]
|
||||
gt_x = list(map(int, np.squeeze(gt[1])))
|
||||
gt_y = list(map(int, np.squeeze(gt[3])))
|
||||
gt_seq_str = str(gt[4].tolist()[0])
|
||||
|
||||
local_sigma_table[gt_id, det_id] = sigma_calculation(
|
||||
det_x, det_y, gt_x, gt_y
|
||||
)
|
||||
local_tau_table[gt_id, det_id] = tau_calculation(
|
||||
det_x, det_y, gt_x, gt_y
|
||||
)
|
||||
local_pred_str[det_id] = pred_seq_str
|
||||
local_gt_str[gt_id] = gt_seq_str
|
||||
|
||||
global_sigma = local_sigma_table
|
||||
global_tau = local_tau_table
|
||||
global_pred_str = local_pred_str
|
||||
global_gt_str = local_gt_str
|
||||
|
||||
single_data = {}
|
||||
single_data["sigma"] = global_sigma
|
||||
single_data["global_tau"] = global_tau
|
||||
single_data["global_pred_str"] = global_pred_str
|
||||
single_data["global_gt_str"] = global_gt_str
|
||||
return single_data
|
||||
|
||||
|
||||
def get_socre_B(gt_dir, img_id, pred_dict):
|
||||
allInputs = 1
|
||||
|
||||
def input_reading_mod(pred_dict):
|
||||
"""This helper reads input from txt files"""
|
||||
det = []
|
||||
n = len(pred_dict)
|
||||
for i in range(n):
|
||||
points = pred_dict[i]["points"]
|
||||
text = pred_dict[i]["texts"]
|
||||
point = ",".join(
|
||||
map(
|
||||
str,
|
||||
points.reshape(
|
||||
-1,
|
||||
),
|
||||
)
|
||||
)
|
||||
det.append([point, text])
|
||||
return det
|
||||
|
||||
def gt_reading_mod(gt_dir, gt_id):
|
||||
gt = io.loadmat("%s/poly_gt_img%s.mat" % (gt_dir, gt_id))
|
||||
gt = gt["polygt"]
|
||||
return gt
|
||||
|
||||
def detection_filtering(detections, groundtruths, threshold=0.5):
|
||||
for gt_id, gt in enumerate(groundtruths):
|
||||
if (gt[5] == "#") and (gt[1].shape[1] > 1):
|
||||
gt_x = list(map(int, np.squeeze(gt[1])))
|
||||
gt_y = list(map(int, np.squeeze(gt[3])))
|
||||
for det_id, detection in enumerate(detections):
|
||||
detection_orig = detection
|
||||
detection = [float(x) for x in detection[0].split(",")]
|
||||
detection = list(map(int, detection))
|
||||
det_x = detection[0::2]
|
||||
det_y = detection[1::2]
|
||||
det_gt_iou = iod(det_x, det_y, gt_x, gt_y)
|
||||
if det_gt_iou > threshold:
|
||||
detections[det_id] = []
|
||||
|
||||
detections[:] = [item for item in detections if item != []]
|
||||
return detections
|
||||
|
||||
def sigma_calculation(det_x, det_y, gt_x, gt_y):
|
||||
"""
|
||||
sigma = inter_area / gt_area
|
||||
"""
|
||||
return np.round(
|
||||
(area_of_intersection(det_x, det_y, gt_x, gt_y) / area(gt_x, gt_y)), 2
|
||||
)
|
||||
|
||||
def tau_calculation(det_x, det_y, gt_x, gt_y):
|
||||
if area(det_x, det_y) == 0.0:
|
||||
return 0
|
||||
return np.round(
|
||||
(area_of_intersection(det_x, det_y, gt_x, gt_y) / area(det_x, det_y)), 2
|
||||
)
|
||||
|
||||
##############################Initialization###################################
|
||||
# global_sigma = []
|
||||
# global_tau = []
|
||||
# global_pred_str = []
|
||||
# global_gt_str = []
|
||||
###############################################################################
|
||||
|
||||
for input_id in range(allInputs):
|
||||
if (
|
||||
(input_id != ".DS_Store")
|
||||
and (input_id != "Pascal_result.txt")
|
||||
and (input_id != "Pascal_result_curved.txt")
|
||||
and (input_id != "Pascal_result_non_curved.txt")
|
||||
and (input_id != "Deteval_result.txt")
|
||||
and (input_id != "Deteval_result_curved.txt")
|
||||
and (input_id != "Deteval_result_non_curved.txt")
|
||||
):
|
||||
detections = input_reading_mod(pred_dict)
|
||||
groundtruths = gt_reading_mod(gt_dir, img_id).tolist()
|
||||
detections = detection_filtering(
|
||||
detections, groundtruths
|
||||
) # filters detections overlapping with DC area
|
||||
dc_id = []
|
||||
for i in range(len(groundtruths)):
|
||||
if groundtruths[i][5] == "#":
|
||||
dc_id.append(i)
|
||||
cnt = 0
|
||||
for a in dc_id:
|
||||
num = a - cnt
|
||||
del groundtruths[num]
|
||||
cnt += 1
|
||||
|
||||
local_sigma_table = np.zeros((len(groundtruths), len(detections)))
|
||||
local_tau_table = np.zeros((len(groundtruths), len(detections)))
|
||||
local_pred_str = {}
|
||||
local_gt_str = {}
|
||||
|
||||
for gt_id, gt in enumerate(groundtruths):
|
||||
if len(detections) > 0:
|
||||
for det_id, detection in enumerate(detections):
|
||||
detection_orig = detection
|
||||
detection = [float(x) for x in detection[0].split(",")]
|
||||
detection = list(map(int, detection))
|
||||
pred_seq_str = detection_orig[1].strip()
|
||||
det_x = detection[0::2]
|
||||
det_y = detection[1::2]
|
||||
gt_x = list(map(int, np.squeeze(gt[1])))
|
||||
gt_y = list(map(int, np.squeeze(gt[3])))
|
||||
gt_seq_str = str(gt[4].tolist()[0])
|
||||
|
||||
local_sigma_table[gt_id, det_id] = sigma_calculation(
|
||||
det_x, det_y, gt_x, gt_y
|
||||
)
|
||||
local_tau_table[gt_id, det_id] = tau_calculation(
|
||||
det_x, det_y, gt_x, gt_y
|
||||
)
|
||||
local_pred_str[det_id] = pred_seq_str
|
||||
local_gt_str[gt_id] = gt_seq_str
|
||||
|
||||
global_sigma = local_sigma_table
|
||||
global_tau = local_tau_table
|
||||
global_pred_str = local_pred_str
|
||||
global_gt_str = local_gt_str
|
||||
|
||||
single_data = {}
|
||||
single_data["sigma"] = global_sigma
|
||||
single_data["global_tau"] = global_tau
|
||||
single_data["global_pred_str"] = global_pred_str
|
||||
single_data["global_gt_str"] = global_gt_str
|
||||
return single_data
|
||||
|
||||
|
||||
def get_score_C(gt_label, text, pred_bboxes):
|
||||
"""
|
||||
get score for CentripetalText (CT) prediction.
|
||||
"""
|
||||
check_install("Polygon", "Polygon3")
|
||||
import Polygon as plg
|
||||
|
||||
def gt_reading_mod(gt_label, text):
|
||||
"""This helper reads groundtruths from mat files"""
|
||||
groundtruths = []
|
||||
nbox = len(gt_label)
|
||||
for i in range(nbox):
|
||||
label = {"transcription": text[i][0], "points": gt_label[i].numpy()}
|
||||
groundtruths.append(label)
|
||||
|
||||
return groundtruths
|
||||
|
||||
def get_union(pD, pG):
|
||||
areaA = pD.area()
|
||||
areaB = pG.area()
|
||||
return areaA + areaB - get_intersection(pD, pG)
|
||||
|
||||
def get_intersection(pD, pG):
|
||||
pInt = pD & pG
|
||||
if len(pInt) == 0:
|
||||
return 0
|
||||
return pInt.area()
|
||||
|
||||
def detection_filtering(detections, groundtruths, threshold=0.5):
|
||||
for gt in groundtruths:
|
||||
point_num = gt["points"].shape[1] // 2
|
||||
if gt["transcription"] == "###" and (point_num > 1):
|
||||
gt_p = np.array(gt["points"]).reshape(point_num, 2).astype("int32")
|
||||
gt_p = plg.Polygon(gt_p)
|
||||
|
||||
for det_id, detection in enumerate(detections):
|
||||
det_y = detection[0::2]
|
||||
det_x = detection[1::2]
|
||||
|
||||
det_p = np.concatenate((np.array(det_x), np.array(det_y)))
|
||||
det_p = det_p.reshape(2, -1).transpose()
|
||||
det_p = plg.Polygon(det_p)
|
||||
|
||||
try:
|
||||
det_gt_iou = get_intersection(det_p, gt_p) / det_p.area()
|
||||
except:
|
||||
print(det_x, det_y, gt_p)
|
||||
if det_gt_iou > threshold:
|
||||
detections[det_id] = []
|
||||
|
||||
detections[:] = [item for item in detections if item != []]
|
||||
return detections
|
||||
|
||||
def sigma_calculation(det_p, gt_p):
|
||||
"""
|
||||
sigma = inter_area / gt_area
|
||||
"""
|
||||
if gt_p.area() == 0.0:
|
||||
return 0
|
||||
return get_intersection(det_p, gt_p) / gt_p.area()
|
||||
|
||||
def tau_calculation(det_p, gt_p):
|
||||
"""
|
||||
tau = inter_area / det_area
|
||||
"""
|
||||
if det_p.area() == 0.0:
|
||||
return 0
|
||||
return get_intersection(det_p, gt_p) / det_p.area()
|
||||
|
||||
detections = []
|
||||
|
||||
for item in pred_bboxes:
|
||||
detections.append(item[:, ::-1].reshape(-1))
|
||||
|
||||
groundtruths = gt_reading_mod(gt_label, text)
|
||||
|
||||
detections = detection_filtering(
|
||||
detections, groundtruths
|
||||
) # filters detections overlapping with DC area
|
||||
|
||||
for idx in range(len(groundtruths) - 1, -1, -1):
|
||||
# NOTE: source code use 'orin' to indicate '#', here we use 'anno',
|
||||
# which may cause slight drop in fscore, about 0.12
|
||||
if groundtruths[idx]["transcription"] == "###":
|
||||
groundtruths.pop(idx)
|
||||
|
||||
local_sigma_table = np.zeros((len(groundtruths), len(detections)))
|
||||
local_tau_table = np.zeros((len(groundtruths), len(detections)))
|
||||
|
||||
for gt_id, gt in enumerate(groundtruths):
|
||||
if len(detections) > 0:
|
||||
for det_id, detection in enumerate(detections):
|
||||
point_num = gt["points"].shape[1] // 2
|
||||
|
||||
gt_p = np.array(gt["points"]).reshape(point_num, 2).astype("int32")
|
||||
gt_p = plg.Polygon(gt_p)
|
||||
|
||||
det_y = detection[0::2]
|
||||
det_x = detection[1::2]
|
||||
|
||||
det_p = np.concatenate((np.array(det_x), np.array(det_y)))
|
||||
|
||||
det_p = det_p.reshape(2, -1).transpose()
|
||||
det_p = plg.Polygon(det_p)
|
||||
|
||||
local_sigma_table[gt_id, det_id] = sigma_calculation(det_p, gt_p)
|
||||
local_tau_table[gt_id, det_id] = tau_calculation(det_p, gt_p)
|
||||
|
||||
data = {}
|
||||
data["sigma"] = local_sigma_table
|
||||
data["global_tau"] = local_tau_table
|
||||
data["global_pred_str"] = ""
|
||||
data["global_gt_str"] = ""
|
||||
return data
|
||||
|
||||
|
||||
def combine_results(all_data, rec_flag=True):
|
||||
tr = 0.7
|
||||
tp = 0.6
|
||||
fsc_k = 0.8
|
||||
k = 2
|
||||
global_sigma = []
|
||||
global_tau = []
|
||||
global_pred_str = []
|
||||
global_gt_str = []
|
||||
|
||||
for data in all_data:
|
||||
global_sigma.append(data["sigma"])
|
||||
global_tau.append(data["global_tau"])
|
||||
global_pred_str.append(data["global_pred_str"])
|
||||
global_gt_str.append(data["global_gt_str"])
|
||||
|
||||
global_accumulative_recall = 0
|
||||
global_accumulative_precision = 0
|
||||
total_num_gt = 0
|
||||
total_num_det = 0
|
||||
hit_str_count = 0
|
||||
hit_count = 0
|
||||
|
||||
def one_to_one(
|
||||
local_sigma_table,
|
||||
local_tau_table,
|
||||
local_accumulative_recall,
|
||||
local_accumulative_precision,
|
||||
global_accumulative_recall,
|
||||
global_accumulative_precision,
|
||||
gt_flag,
|
||||
det_flag,
|
||||
idy,
|
||||
rec_flag,
|
||||
):
|
||||
hit_str_num = 0
|
||||
for gt_id in range(num_gt):
|
||||
gt_matching_qualified_sigma_candidates = np.where(
|
||||
local_sigma_table[gt_id, :] > tr
|
||||
)
|
||||
gt_matching_num_qualified_sigma_candidates = (
|
||||
gt_matching_qualified_sigma_candidates[0].shape[0]
|
||||
)
|
||||
gt_matching_qualified_tau_candidates = np.where(
|
||||
local_tau_table[gt_id, :] > tp
|
||||
)
|
||||
gt_matching_num_qualified_tau_candidates = (
|
||||
gt_matching_qualified_tau_candidates[0].shape[0]
|
||||
)
|
||||
|
||||
det_matching_qualified_sigma_candidates = np.where(
|
||||
local_sigma_table[:, gt_matching_qualified_sigma_candidates[0]] > tr
|
||||
)
|
||||
det_matching_num_qualified_sigma_candidates = (
|
||||
det_matching_qualified_sigma_candidates[0].shape[0]
|
||||
)
|
||||
det_matching_qualified_tau_candidates = np.where(
|
||||
local_tau_table[:, gt_matching_qualified_tau_candidates[0]] > tp
|
||||
)
|
||||
det_matching_num_qualified_tau_candidates = (
|
||||
det_matching_qualified_tau_candidates[0].shape[0]
|
||||
)
|
||||
|
||||
if (
|
||||
(gt_matching_num_qualified_sigma_candidates == 1)
|
||||
and (gt_matching_num_qualified_tau_candidates == 1)
|
||||
and (det_matching_num_qualified_sigma_candidates == 1)
|
||||
and (det_matching_num_qualified_tau_candidates == 1)
|
||||
):
|
||||
global_accumulative_recall = global_accumulative_recall + 1.0
|
||||
global_accumulative_precision = global_accumulative_precision + 1.0
|
||||
local_accumulative_recall = local_accumulative_recall + 1.0
|
||||
local_accumulative_precision = local_accumulative_precision + 1.0
|
||||
|
||||
gt_flag[0, gt_id] = 1
|
||||
matched_det_id = np.where(local_sigma_table[gt_id, :] > tr)
|
||||
# recg start
|
||||
if rec_flag:
|
||||
gt_str_cur = global_gt_str[idy][gt_id]
|
||||
pred_str_cur = global_pred_str[idy][matched_det_id[0].tolist()[0]]
|
||||
if pred_str_cur == gt_str_cur:
|
||||
hit_str_num += 1
|
||||
else:
|
||||
if pred_str_cur.lower() == gt_str_cur.lower():
|
||||
hit_str_num += 1
|
||||
# recg end
|
||||
det_flag[0, matched_det_id] = 1
|
||||
return (
|
||||
local_accumulative_recall,
|
||||
local_accumulative_precision,
|
||||
global_accumulative_recall,
|
||||
global_accumulative_precision,
|
||||
gt_flag,
|
||||
det_flag,
|
||||
hit_str_num,
|
||||
)
|
||||
|
||||
def one_to_many(
|
||||
local_sigma_table,
|
||||
local_tau_table,
|
||||
local_accumulative_recall,
|
||||
local_accumulative_precision,
|
||||
global_accumulative_recall,
|
||||
global_accumulative_precision,
|
||||
gt_flag,
|
||||
det_flag,
|
||||
idy,
|
||||
rec_flag,
|
||||
):
|
||||
hit_str_num = 0
|
||||
for gt_id in range(num_gt):
|
||||
# skip the following if the groundtruth was matched
|
||||
if gt_flag[0, gt_id] > 0:
|
||||
continue
|
||||
|
||||
non_zero_in_sigma = np.where(local_sigma_table[gt_id, :] > 0)
|
||||
num_non_zero_in_sigma = non_zero_in_sigma[0].shape[0]
|
||||
|
||||
if num_non_zero_in_sigma >= k:
|
||||
####search for all detections that overlaps with this groundtruth
|
||||
qualified_tau_candidates = np.where(
|
||||
(local_tau_table[gt_id, :] >= tp) & (det_flag[0, :] == 0)
|
||||
)
|
||||
num_qualified_tau_candidates = qualified_tau_candidates[0].shape[0]
|
||||
|
||||
if num_qualified_tau_candidates == 1:
|
||||
if (local_tau_table[gt_id, qualified_tau_candidates] >= tp) and (
|
||||
local_sigma_table[gt_id, qualified_tau_candidates] >= tr
|
||||
):
|
||||
# became an one-to-one case
|
||||
global_accumulative_recall = global_accumulative_recall + 1.0
|
||||
global_accumulative_precision = (
|
||||
global_accumulative_precision + 1.0
|
||||
)
|
||||
local_accumulative_recall = local_accumulative_recall + 1.0
|
||||
local_accumulative_precision = (
|
||||
local_accumulative_precision + 1.0
|
||||
)
|
||||
|
||||
gt_flag[0, gt_id] = 1
|
||||
det_flag[0, qualified_tau_candidates] = 1
|
||||
# recg start
|
||||
if rec_flag:
|
||||
gt_str_cur = global_gt_str[idy][gt_id]
|
||||
pred_str_cur = global_pred_str[idy][
|
||||
qualified_tau_candidates[0].tolist()[0]
|
||||
]
|
||||
if pred_str_cur == gt_str_cur:
|
||||
hit_str_num += 1
|
||||
else:
|
||||
if pred_str_cur.lower() == gt_str_cur.lower():
|
||||
hit_str_num += 1
|
||||
# recg end
|
||||
elif np.sum(local_sigma_table[gt_id, qualified_tau_candidates]) >= tr:
|
||||
gt_flag[0, gt_id] = 1
|
||||
det_flag[0, qualified_tau_candidates] = 1
|
||||
# recg start
|
||||
if rec_flag:
|
||||
gt_str_cur = global_gt_str[idy][gt_id]
|
||||
pred_str_cur = global_pred_str[idy][
|
||||
qualified_tau_candidates[0].tolist()[0]
|
||||
]
|
||||
if pred_str_cur == gt_str_cur:
|
||||
hit_str_num += 1
|
||||
else:
|
||||
if pred_str_cur.lower() == gt_str_cur.lower():
|
||||
hit_str_num += 1
|
||||
# recg end
|
||||
|
||||
global_accumulative_recall = global_accumulative_recall + fsc_k
|
||||
global_accumulative_precision = (
|
||||
global_accumulative_precision
|
||||
+ num_qualified_tau_candidates * fsc_k
|
||||
)
|
||||
|
||||
local_accumulative_recall = local_accumulative_recall + fsc_k
|
||||
local_accumulative_precision = (
|
||||
local_accumulative_precision
|
||||
+ num_qualified_tau_candidates * fsc_k
|
||||
)
|
||||
|
||||
return (
|
||||
local_accumulative_recall,
|
||||
local_accumulative_precision,
|
||||
global_accumulative_recall,
|
||||
global_accumulative_precision,
|
||||
gt_flag,
|
||||
det_flag,
|
||||
hit_str_num,
|
||||
)
|
||||
|
||||
def many_to_one(
|
||||
local_sigma_table,
|
||||
local_tau_table,
|
||||
local_accumulative_recall,
|
||||
local_accumulative_precision,
|
||||
global_accumulative_recall,
|
||||
global_accumulative_precision,
|
||||
gt_flag,
|
||||
det_flag,
|
||||
idy,
|
||||
rec_flag,
|
||||
):
|
||||
hit_str_num = 0
|
||||
for det_id in range(num_det):
|
||||
# skip the following if the detection was matched
|
||||
if det_flag[0, det_id] > 0:
|
||||
continue
|
||||
|
||||
non_zero_in_tau = np.where(local_tau_table[:, det_id] > 0)
|
||||
num_non_zero_in_tau = non_zero_in_tau[0].shape[0]
|
||||
|
||||
if num_non_zero_in_tau >= k:
|
||||
####search for all detections that overlaps with this groundtruth
|
||||
qualified_sigma_candidates = np.where(
|
||||
(local_sigma_table[:, det_id] >= tp) & (gt_flag[0, :] == 0)
|
||||
)
|
||||
num_qualified_sigma_candidates = qualified_sigma_candidates[0].shape[0]
|
||||
|
||||
if num_qualified_sigma_candidates == 1:
|
||||
if (local_tau_table[qualified_sigma_candidates, det_id] >= tp) and (
|
||||
local_sigma_table[qualified_sigma_candidates, det_id] >= tr
|
||||
):
|
||||
# became an one-to-one case
|
||||
global_accumulative_recall = global_accumulative_recall + 1.0
|
||||
global_accumulative_precision = (
|
||||
global_accumulative_precision + 1.0
|
||||
)
|
||||
local_accumulative_recall = local_accumulative_recall + 1.0
|
||||
local_accumulative_precision = (
|
||||
local_accumulative_precision + 1.0
|
||||
)
|
||||
|
||||
gt_flag[0, qualified_sigma_candidates] = 1
|
||||
det_flag[0, det_id] = 1
|
||||
# recg start
|
||||
if rec_flag:
|
||||
pred_str_cur = global_pred_str[idy][det_id]
|
||||
gt_len = len(qualified_sigma_candidates[0])
|
||||
for idx in range(gt_len):
|
||||
ele_gt_id = qualified_sigma_candidates[0].tolist()[idx]
|
||||
if ele_gt_id not in global_gt_str[idy]:
|
||||
continue
|
||||
gt_str_cur = global_gt_str[idy][ele_gt_id]
|
||||
if pred_str_cur == gt_str_cur:
|
||||
hit_str_num += 1
|
||||
break
|
||||
else:
|
||||
if pred_str_cur.lower() == gt_str_cur.lower():
|
||||
hit_str_num += 1
|
||||
break
|
||||
# recg end
|
||||
elif np.sum(local_tau_table[qualified_sigma_candidates, det_id]) >= tp:
|
||||
det_flag[0, det_id] = 1
|
||||
gt_flag[0, qualified_sigma_candidates] = 1
|
||||
# recg start
|
||||
if rec_flag:
|
||||
pred_str_cur = global_pred_str[idy][det_id]
|
||||
gt_len = len(qualified_sigma_candidates[0])
|
||||
for idx in range(gt_len):
|
||||
ele_gt_id = qualified_sigma_candidates[0].tolist()[idx]
|
||||
if ele_gt_id not in global_gt_str[idy]:
|
||||
continue
|
||||
gt_str_cur = global_gt_str[idy][ele_gt_id]
|
||||
if pred_str_cur == gt_str_cur:
|
||||
hit_str_num += 1
|
||||
break
|
||||
else:
|
||||
if pred_str_cur.lower() == gt_str_cur.lower():
|
||||
hit_str_num += 1
|
||||
break
|
||||
# recg end
|
||||
|
||||
global_accumulative_recall = (
|
||||
global_accumulative_recall
|
||||
+ num_qualified_sigma_candidates * fsc_k
|
||||
)
|
||||
global_accumulative_precision = (
|
||||
global_accumulative_precision + fsc_k
|
||||
)
|
||||
|
||||
local_accumulative_recall = (
|
||||
local_accumulative_recall
|
||||
+ num_qualified_sigma_candidates * fsc_k
|
||||
)
|
||||
local_accumulative_precision = local_accumulative_precision + fsc_k
|
||||
return (
|
||||
local_accumulative_recall,
|
||||
local_accumulative_precision,
|
||||
global_accumulative_recall,
|
||||
global_accumulative_precision,
|
||||
gt_flag,
|
||||
det_flag,
|
||||
hit_str_num,
|
||||
)
|
||||
|
||||
for idx in range(len(global_sigma)):
|
||||
local_sigma_table = np.array(global_sigma[idx])
|
||||
local_tau_table = global_tau[idx]
|
||||
|
||||
num_gt = local_sigma_table.shape[0]
|
||||
num_det = local_sigma_table.shape[1]
|
||||
|
||||
total_num_gt = total_num_gt + num_gt
|
||||
total_num_det = total_num_det + num_det
|
||||
|
||||
local_accumulative_recall = 0
|
||||
local_accumulative_precision = 0
|
||||
gt_flag = np.zeros((1, num_gt))
|
||||
det_flag = np.zeros((1, num_det))
|
||||
|
||||
#######first check for one-to-one case##########
|
||||
(
|
||||
local_accumulative_recall,
|
||||
local_accumulative_precision,
|
||||
global_accumulative_recall,
|
||||
global_accumulative_precision,
|
||||
gt_flag,
|
||||
det_flag,
|
||||
hit_str_num,
|
||||
) = one_to_one(
|
||||
local_sigma_table,
|
||||
local_tau_table,
|
||||
local_accumulative_recall,
|
||||
local_accumulative_precision,
|
||||
global_accumulative_recall,
|
||||
global_accumulative_precision,
|
||||
gt_flag,
|
||||
det_flag,
|
||||
idx,
|
||||
rec_flag,
|
||||
)
|
||||
|
||||
hit_str_count += hit_str_num
|
||||
#######then check for one-to-many case##########
|
||||
(
|
||||
local_accumulative_recall,
|
||||
local_accumulative_precision,
|
||||
global_accumulative_recall,
|
||||
global_accumulative_precision,
|
||||
gt_flag,
|
||||
det_flag,
|
||||
hit_str_num,
|
||||
) = one_to_many(
|
||||
local_sigma_table,
|
||||
local_tau_table,
|
||||
local_accumulative_recall,
|
||||
local_accumulative_precision,
|
||||
global_accumulative_recall,
|
||||
global_accumulative_precision,
|
||||
gt_flag,
|
||||
det_flag,
|
||||
idx,
|
||||
rec_flag,
|
||||
)
|
||||
hit_str_count += hit_str_num
|
||||
#######then check for many-to-one case##########
|
||||
(
|
||||
local_accumulative_recall,
|
||||
local_accumulative_precision,
|
||||
global_accumulative_recall,
|
||||
global_accumulative_precision,
|
||||
gt_flag,
|
||||
det_flag,
|
||||
hit_str_num,
|
||||
) = many_to_one(
|
||||
local_sigma_table,
|
||||
local_tau_table,
|
||||
local_accumulative_recall,
|
||||
local_accumulative_precision,
|
||||
global_accumulative_recall,
|
||||
global_accumulative_precision,
|
||||
gt_flag,
|
||||
det_flag,
|
||||
idx,
|
||||
rec_flag,
|
||||
)
|
||||
hit_str_count += hit_str_num
|
||||
|
||||
try:
|
||||
recall = global_accumulative_recall / total_num_gt
|
||||
except ZeroDivisionError:
|
||||
recall = 0
|
||||
|
||||
try:
|
||||
precision = global_accumulative_precision / total_num_det
|
||||
except ZeroDivisionError:
|
||||
precision = 0
|
||||
|
||||
try:
|
||||
f_score = 2 * precision * recall / (precision + recall)
|
||||
except ZeroDivisionError:
|
||||
f_score = 0
|
||||
|
||||
try:
|
||||
seqerr = 1 - float(hit_str_count) / global_accumulative_recall
|
||||
except ZeroDivisionError:
|
||||
seqerr = 1
|
||||
|
||||
try:
|
||||
recall_e2e = float(hit_str_count) / total_num_gt
|
||||
except ZeroDivisionError:
|
||||
recall_e2e = 0
|
||||
|
||||
try:
|
||||
precision_e2e = float(hit_str_count) / total_num_det
|
||||
except ZeroDivisionError:
|
||||
precision_e2e = 0
|
||||
|
||||
try:
|
||||
f_score_e2e = 2 * precision_e2e * recall_e2e / (precision_e2e + recall_e2e)
|
||||
except ZeroDivisionError:
|
||||
f_score_e2e = 0
|
||||
|
||||
final = {
|
||||
"total_num_gt": total_num_gt,
|
||||
"total_num_det": total_num_det,
|
||||
"global_accumulative_recall": global_accumulative_recall,
|
||||
"hit_str_count": hit_str_count,
|
||||
"recall": recall,
|
||||
"precision": precision,
|
||||
"f_score": f_score,
|
||||
"seqerr": seqerr,
|
||||
"recall_e2e": recall_e2e,
|
||||
"precision_e2e": precision_e2e,
|
||||
"f_score_e2e": f_score_e2e,
|
||||
}
|
||||
return final
|
||||
84
ppocr/utils/e2e_metric/polygon_fast.py
Executable file
84
ppocr/utils/e2e_metric/polygon_fast.py
Executable file
@@ -0,0 +1,84 @@
|
||||
# Copyright (c) 2021 PaddlePaddle Authors. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
import numpy as np
|
||||
from shapely.geometry import Polygon
|
||||
|
||||
"""
|
||||
:param det_x: [1, N] Xs of detection's vertices
|
||||
:param det_y: [1, N] Ys of detection's vertices
|
||||
:param gt_x: [1, N] Xs of groundtruth's vertices
|
||||
:param gt_y: [1, N] Ys of groundtruth's vertices
|
||||
|
||||
##############
|
||||
All the calculation of 'AREA' in this script is handled by:
|
||||
1) First generating a binary mask with the polygon area filled up with 1's
|
||||
2) Summing up all the 1's
|
||||
"""
|
||||
|
||||
|
||||
def area(x, y):
|
||||
polygon = Polygon(np.stack([x, y], axis=1))
|
||||
return float(polygon.area)
|
||||
|
||||
|
||||
def approx_area_of_intersection(det_x, det_y, gt_x, gt_y):
|
||||
"""
|
||||
This helper determine if both polygons are intersecting with each others with an approximation method.
|
||||
Area of intersection represented by the minimum bounding rectangular [xmin, ymin, xmax, ymax]
|
||||
"""
|
||||
det_ymax = np.max(det_y)
|
||||
det_xmax = np.max(det_x)
|
||||
det_ymin = np.min(det_y)
|
||||
det_xmin = np.min(det_x)
|
||||
|
||||
gt_ymax = np.max(gt_y)
|
||||
gt_xmax = np.max(gt_x)
|
||||
gt_ymin = np.min(gt_y)
|
||||
gt_xmin = np.min(gt_x)
|
||||
|
||||
all_min_ymax = np.minimum(det_ymax, gt_ymax)
|
||||
all_max_ymin = np.maximum(det_ymin, gt_ymin)
|
||||
|
||||
intersect_heights = np.maximum(0.0, (all_min_ymax - all_max_ymin))
|
||||
|
||||
all_min_xmax = np.minimum(det_xmax, gt_xmax)
|
||||
all_max_xmin = np.maximum(det_xmin, gt_xmin)
|
||||
intersect_widths = np.maximum(0.0, (all_min_xmax - all_max_xmin))
|
||||
|
||||
return intersect_heights * intersect_widths
|
||||
|
||||
|
||||
def area_of_intersection(det_x, det_y, gt_x, gt_y):
|
||||
p1 = Polygon(np.stack([det_x, det_y], axis=1)).buffer(0)
|
||||
p2 = Polygon(np.stack([gt_x, gt_y], axis=1)).buffer(0)
|
||||
return float(p1.intersection(p2).area)
|
||||
|
||||
|
||||
def area_of_union(det_x, det_y, gt_x, gt_y):
|
||||
p1 = Polygon(np.stack([det_x, det_y], axis=1)).buffer(0)
|
||||
p2 = Polygon(np.stack([gt_x, gt_y], axis=1)).buffer(0)
|
||||
return float(p1.union(p2).area)
|
||||
|
||||
|
||||
def iou(det_x, det_y, gt_x, gt_y):
|
||||
return area_of_intersection(det_x, det_y, gt_x, gt_y) / (
|
||||
area_of_union(det_x, det_y, gt_x, gt_y) + 1.0
|
||||
)
|
||||
|
||||
|
||||
def iod(det_x, det_y, gt_x, gt_y):
|
||||
"""
|
||||
This helper determine the fraction of intersection area over detection area
|
||||
"""
|
||||
return area_of_intersection(det_x, det_y, gt_x, gt_y) / (area(det_x, det_y) + 1.0)
|
||||
88
ppocr/utils/e2e_utils/extract_batchsize.py
Normal file
88
ppocr/utils/e2e_utils/extract_batchsize.py
Normal file
@@ -0,0 +1,88 @@
|
||||
import paddle
|
||||
import numpy as np
|
||||
import copy
|
||||
|
||||
|
||||
def org_tcl_rois(batch_size, pos_lists, pos_masks, label_lists, tcl_bs):
|
||||
""" """
|
||||
pos_lists_, pos_masks_, label_lists_ = [], [], []
|
||||
img_bs = batch_size
|
||||
ngpu = int(batch_size / img_bs)
|
||||
img_ids = np.array(pos_lists, dtype=np.int32)[:, 0, 0].copy()
|
||||
pos_lists_split, pos_masks_split, label_lists_split = [], [], []
|
||||
for i in range(ngpu):
|
||||
pos_lists_split.append([])
|
||||
pos_masks_split.append([])
|
||||
label_lists_split.append([])
|
||||
|
||||
for i in range(img_ids.shape[0]):
|
||||
img_id = img_ids[i]
|
||||
gpu_id = int(img_id / img_bs)
|
||||
img_id = img_id % img_bs
|
||||
pos_list = pos_lists[i].copy()
|
||||
pos_list[:, 0] = img_id
|
||||
pos_lists_split[gpu_id].append(pos_list)
|
||||
pos_masks_split[gpu_id].append(pos_masks[i].copy())
|
||||
label_lists_split[gpu_id].append(copy.deepcopy(label_lists[i]))
|
||||
# repeat or delete
|
||||
for i in range(ngpu):
|
||||
vp_len = len(pos_lists_split[i])
|
||||
if vp_len <= tcl_bs:
|
||||
for j in range(0, tcl_bs - vp_len):
|
||||
pos_list = pos_lists_split[i][j].copy()
|
||||
pos_lists_split[i].append(pos_list)
|
||||
pos_mask = pos_masks_split[i][j].copy()
|
||||
pos_masks_split[i].append(pos_mask)
|
||||
label_list = copy.deepcopy(label_lists_split[i][j])
|
||||
label_lists_split[i].append(label_list)
|
||||
else:
|
||||
for j in range(0, vp_len - tcl_bs):
|
||||
c_len = len(pos_lists_split[i])
|
||||
pop_id = np.random.permutation(c_len)[0]
|
||||
pos_lists_split[i].pop(pop_id)
|
||||
pos_masks_split[i].pop(pop_id)
|
||||
label_lists_split[i].pop(pop_id)
|
||||
# merge
|
||||
for i in range(ngpu):
|
||||
pos_lists_.extend(pos_lists_split[i])
|
||||
pos_masks_.extend(pos_masks_split[i])
|
||||
label_lists_.extend(label_lists_split[i])
|
||||
return pos_lists_, pos_masks_, label_lists_
|
||||
|
||||
|
||||
def pre_process(
|
||||
label_list, pos_list, pos_mask, max_text_length, max_text_nums, pad_num, tcl_bs
|
||||
):
|
||||
label_list = label_list.numpy()
|
||||
batch, _, _, _ = label_list.shape
|
||||
pos_list = pos_list.numpy()
|
||||
pos_mask = pos_mask.numpy()
|
||||
pos_list_t = []
|
||||
pos_mask_t = []
|
||||
label_list_t = []
|
||||
for i in range(batch):
|
||||
for j in range(max_text_nums):
|
||||
if pos_mask[i, j].any():
|
||||
pos_list_t.append(pos_list[i][j])
|
||||
pos_mask_t.append(pos_mask[i][j])
|
||||
label_list_t.append(label_list[i][j])
|
||||
pos_list, pos_mask, label_list = org_tcl_rois(
|
||||
batch, pos_list_t, pos_mask_t, label_list_t, tcl_bs
|
||||
)
|
||||
label = []
|
||||
tt = [l.tolist() for l in label_list]
|
||||
for i in range(tcl_bs):
|
||||
k = 0
|
||||
for j in range(max_text_length):
|
||||
if tt[i][j][0] != pad_num:
|
||||
k += 1
|
||||
else:
|
||||
break
|
||||
label.append(k)
|
||||
label = paddle.to_tensor(label)
|
||||
label = paddle.cast(label, dtype="int64")
|
||||
pos_list = paddle.to_tensor(pos_list)
|
||||
pos_mask = paddle.to_tensor(pos_mask)
|
||||
label_list = paddle.squeeze(paddle.to_tensor(label_list), axis=2)
|
||||
label_list = paddle.cast(label_list, dtype="int32")
|
||||
return pos_list, pos_mask, label_list, label
|
||||
523
ppocr/utils/e2e_utils/extract_textpoint_fast.py
Normal file
523
ppocr/utils/e2e_utils/extract_textpoint_fast.py
Normal file
@@ -0,0 +1,523 @@
|
||||
# Copyright (c) 2021 PaddlePaddle Authors. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""Contains various CTC decoders."""
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import cv2
|
||||
import math
|
||||
|
||||
import numpy as np
|
||||
from itertools import groupby
|
||||
from skimage.morphology._skeletonize import thin
|
||||
|
||||
|
||||
def get_dict(character_dict_path):
|
||||
character_str = ""
|
||||
with open(character_dict_path, "rb") as fin:
|
||||
lines = fin.readlines()
|
||||
for line in lines:
|
||||
line = line.decode("utf-8").strip("\n").strip("\r\n")
|
||||
character_str += line
|
||||
dict_character = list(character_str)
|
||||
return dict_character
|
||||
|
||||
|
||||
def softmax(logits):
|
||||
"""
|
||||
logits: N x d
|
||||
"""
|
||||
max_value = np.max(logits, axis=1, keepdims=True)
|
||||
exp = np.exp(logits - max_value)
|
||||
exp_sum = np.sum(exp, axis=1, keepdims=True)
|
||||
dist = exp / exp_sum
|
||||
return dist
|
||||
|
||||
|
||||
def get_keep_pos_idxs(labels, remove_blank=None):
|
||||
"""
|
||||
Remove duplicate and get pos idxs of keep items.
|
||||
The value of keep_blank should be [None, 95].
|
||||
"""
|
||||
duplicate_len_list = []
|
||||
keep_pos_idx_list = []
|
||||
keep_char_idx_list = []
|
||||
for k, v_ in groupby(labels):
|
||||
current_len = len(list(v_))
|
||||
if k != remove_blank:
|
||||
current_idx = int(sum(duplicate_len_list) + current_len // 2)
|
||||
keep_pos_idx_list.append(current_idx)
|
||||
keep_char_idx_list.append(k)
|
||||
duplicate_len_list.append(current_len)
|
||||
return keep_char_idx_list, keep_pos_idx_list
|
||||
|
||||
|
||||
def remove_blank(labels, blank=0):
|
||||
new_labels = [x for x in labels if x != blank]
|
||||
return new_labels
|
||||
|
||||
|
||||
def insert_blank(labels, blank=0):
|
||||
new_labels = [blank]
|
||||
for l in labels:
|
||||
new_labels += [l, blank]
|
||||
return new_labels
|
||||
|
||||
|
||||
def ctc_greedy_decoder(probs_seq, blank=95, keep_blank_in_idxs=True):
|
||||
"""
|
||||
CTC greedy (best path) decoder.
|
||||
"""
|
||||
raw_str = np.argmax(np.array(probs_seq), axis=1)
|
||||
remove_blank_in_pos = None if keep_blank_in_idxs else blank
|
||||
dedup_str, keep_idx_list = get_keep_pos_idxs(
|
||||
raw_str, remove_blank=remove_blank_in_pos
|
||||
)
|
||||
dst_str = remove_blank(dedup_str, blank=blank)
|
||||
return dst_str, keep_idx_list
|
||||
|
||||
|
||||
def instance_ctc_greedy_decoder(
|
||||
gather_info, logits_map, pts_num=4, point_gather_mode=None
|
||||
):
|
||||
_, _, C = logits_map.shape
|
||||
if point_gather_mode == "align":
|
||||
insert_num = 0
|
||||
gather_info = np.array(gather_info)
|
||||
length = len(gather_info) - 1
|
||||
for index in range(length):
|
||||
stride_y = np.abs(
|
||||
gather_info[index + insert_num][0]
|
||||
- gather_info[index + 1 + insert_num][0]
|
||||
)
|
||||
stride_x = np.abs(
|
||||
gather_info[index + insert_num][1]
|
||||
- gather_info[index + 1 + insert_num][1]
|
||||
)
|
||||
max_points = int(max(stride_x, stride_y))
|
||||
stride = (
|
||||
gather_info[index + insert_num] - gather_info[index + 1 + insert_num]
|
||||
) / (max_points)
|
||||
insert_num_temp = max_points - 1
|
||||
|
||||
for i in range(int(insert_num_temp)):
|
||||
insert_value = gather_info[index + insert_num] - (i + 1) * stride
|
||||
insert_index = index + i + 1 + insert_num
|
||||
gather_info = np.insert(gather_info, insert_index, insert_value, axis=0)
|
||||
insert_num += insert_num_temp
|
||||
gather_info = gather_info.tolist()
|
||||
else:
|
||||
pass
|
||||
ys, xs = zip(*gather_info)
|
||||
logits_seq = logits_map[list(ys), list(xs)]
|
||||
probs_seq = logits_seq
|
||||
labels = np.argmax(probs_seq, axis=1)
|
||||
dst_str = [k for k, v_ in groupby(labels) if k != C - 1]
|
||||
detal = len(gather_info) // (pts_num - 1)
|
||||
keep_idx_list = [0] + [detal * (i + 1) for i in range(pts_num - 2)] + [-1]
|
||||
keep_gather_list = [gather_info[idx] for idx in keep_idx_list]
|
||||
return dst_str, keep_gather_list
|
||||
|
||||
|
||||
def ctc_decoder_for_image(
|
||||
gather_info_list, logits_map, Lexicon_Table, pts_num=6, point_gather_mode=None
|
||||
):
|
||||
"""
|
||||
CTC decoder using multiple processes.
|
||||
"""
|
||||
decoder_str = []
|
||||
decoder_xys = []
|
||||
for gather_info in gather_info_list:
|
||||
if len(gather_info) < pts_num:
|
||||
continue
|
||||
dst_str, xys_list = instance_ctc_greedy_decoder(
|
||||
gather_info,
|
||||
logits_map,
|
||||
pts_num=pts_num,
|
||||
point_gather_mode=point_gather_mode,
|
||||
)
|
||||
dst_str_readable = "".join([Lexicon_Table[idx] for idx in dst_str])
|
||||
if len(dst_str_readable) < 2:
|
||||
continue
|
||||
decoder_str.append(dst_str_readable)
|
||||
decoder_xys.append(xys_list)
|
||||
return decoder_str, decoder_xys
|
||||
|
||||
|
||||
def sort_with_direction(pos_list, f_direction):
|
||||
"""
|
||||
f_direction: h x w x 2
|
||||
pos_list: [[y, x], [y, x], [y, x] ...]
|
||||
"""
|
||||
|
||||
def sort_part_with_direction(pos_list, point_direction):
|
||||
pos_list = np.array(pos_list).reshape(-1, 2)
|
||||
point_direction = np.array(point_direction).reshape(-1, 2)
|
||||
average_direction = np.mean(point_direction, axis=0, keepdims=True)
|
||||
pos_proj_leng = np.sum(pos_list * average_direction, axis=1)
|
||||
sorted_list = pos_list[np.argsort(pos_proj_leng)].tolist()
|
||||
sorted_direction = point_direction[np.argsort(pos_proj_leng)].tolist()
|
||||
return sorted_list, sorted_direction
|
||||
|
||||
pos_list = np.array(pos_list).reshape(-1, 2)
|
||||
point_direction = f_direction[pos_list[:, 0], pos_list[:, 1]] # x, y
|
||||
point_direction = point_direction[:, ::-1] # x, y -> y, x
|
||||
sorted_point, sorted_direction = sort_part_with_direction(pos_list, point_direction)
|
||||
|
||||
point_num = len(sorted_point)
|
||||
if point_num >= 16:
|
||||
middle_num = point_num // 2
|
||||
first_part_point = sorted_point[:middle_num]
|
||||
first_point_direction = sorted_direction[:middle_num]
|
||||
sorted_fist_part_point, sorted_fist_part_direction = sort_part_with_direction(
|
||||
first_part_point, first_point_direction
|
||||
)
|
||||
|
||||
last_part_point = sorted_point[middle_num:]
|
||||
last_point_direction = sorted_direction[middle_num:]
|
||||
sorted_last_part_point, sorted_last_part_direction = sort_part_with_direction(
|
||||
last_part_point, last_point_direction
|
||||
)
|
||||
sorted_point = sorted_fist_part_point + sorted_last_part_point
|
||||
sorted_direction = sorted_fist_part_direction + sorted_last_part_direction
|
||||
|
||||
return sorted_point, np.array(sorted_direction)
|
||||
|
||||
|
||||
def add_id(pos_list, image_id=0):
|
||||
"""
|
||||
Add id for gather feature, for inference.
|
||||
"""
|
||||
new_list = []
|
||||
for item in pos_list:
|
||||
new_list.append((image_id, item[0], item[1]))
|
||||
return new_list
|
||||
|
||||
|
||||
def sort_and_expand_with_direction(pos_list, f_direction):
|
||||
"""
|
||||
f_direction: h x w x 2
|
||||
pos_list: [[y, x], [y, x], [y, x] ...]
|
||||
"""
|
||||
h, w, _ = f_direction.shape
|
||||
sorted_list, point_direction = sort_with_direction(pos_list, f_direction)
|
||||
|
||||
point_num = len(sorted_list)
|
||||
sub_direction_len = max(point_num // 3, 2)
|
||||
left_direction = point_direction[:sub_direction_len, :]
|
||||
right_dirction = point_direction[point_num - sub_direction_len :, :]
|
||||
|
||||
left_average_direction = -np.mean(left_direction, axis=0, keepdims=True)
|
||||
left_average_len = np.linalg.norm(left_average_direction)
|
||||
left_start = np.array(sorted_list[0])
|
||||
left_step = left_average_direction / (left_average_len + 1e-6)
|
||||
|
||||
right_average_direction = np.mean(right_dirction, axis=0, keepdims=True)
|
||||
right_average_len = np.linalg.norm(right_average_direction)
|
||||
right_step = right_average_direction / (right_average_len + 1e-6)
|
||||
right_start = np.array(sorted_list[-1])
|
||||
|
||||
append_num = max(int((left_average_len + right_average_len) / 2.0 * 0.15), 1)
|
||||
left_list = []
|
||||
right_list = []
|
||||
for i in range(append_num):
|
||||
ly, lx = (
|
||||
np.round(left_start + left_step * (i + 1))
|
||||
.flatten()
|
||||
.astype("int32")
|
||||
.tolist()
|
||||
)
|
||||
if ly < h and lx < w and (ly, lx) not in left_list:
|
||||
left_list.append((ly, lx))
|
||||
ry, rx = (
|
||||
np.round(right_start + right_step * (i + 1))
|
||||
.flatten()
|
||||
.astype("int32")
|
||||
.tolist()
|
||||
)
|
||||
if ry < h and rx < w and (ry, rx) not in right_list:
|
||||
right_list.append((ry, rx))
|
||||
|
||||
all_list = left_list[::-1] + sorted_list + right_list
|
||||
return all_list
|
||||
|
||||
|
||||
def sort_and_expand_with_direction_v2(pos_list, f_direction, binary_tcl_map):
|
||||
"""
|
||||
f_direction: h x w x 2
|
||||
pos_list: [[y, x], [y, x], [y, x] ...]
|
||||
binary_tcl_map: h x w
|
||||
"""
|
||||
h, w, _ = f_direction.shape
|
||||
sorted_list, point_direction = sort_with_direction(pos_list, f_direction)
|
||||
|
||||
point_num = len(sorted_list)
|
||||
sub_direction_len = max(point_num // 3, 2)
|
||||
left_direction = point_direction[:sub_direction_len, :]
|
||||
right_dirction = point_direction[point_num - sub_direction_len :, :]
|
||||
|
||||
left_average_direction = -np.mean(left_direction, axis=0, keepdims=True)
|
||||
left_average_len = np.linalg.norm(left_average_direction)
|
||||
left_start = np.array(sorted_list[0])
|
||||
left_step = left_average_direction / (left_average_len + 1e-6)
|
||||
|
||||
right_average_direction = np.mean(right_dirction, axis=0, keepdims=True)
|
||||
right_average_len = np.linalg.norm(right_average_direction)
|
||||
right_step = right_average_direction / (right_average_len + 1e-6)
|
||||
right_start = np.array(sorted_list[-1])
|
||||
|
||||
append_num = max(int((left_average_len + right_average_len) / 2.0 * 0.15), 1)
|
||||
max_append_num = 2 * append_num
|
||||
|
||||
left_list = []
|
||||
right_list = []
|
||||
for i in range(max_append_num):
|
||||
ly, lx = (
|
||||
np.round(left_start + left_step * (i + 1))
|
||||
.flatten()
|
||||
.astype("int32")
|
||||
.tolist()
|
||||
)
|
||||
if ly < h and lx < w and (ly, lx) not in left_list:
|
||||
if binary_tcl_map[ly, lx] > 0.5:
|
||||
left_list.append((ly, lx))
|
||||
else:
|
||||
break
|
||||
|
||||
for i in range(max_append_num):
|
||||
ry, rx = (
|
||||
np.round(right_start + right_step * (i + 1))
|
||||
.flatten()
|
||||
.astype("int32")
|
||||
.tolist()
|
||||
)
|
||||
if ry < h and rx < w and (ry, rx) not in right_list:
|
||||
if binary_tcl_map[ry, rx] > 0.5:
|
||||
right_list.append((ry, rx))
|
||||
else:
|
||||
break
|
||||
|
||||
all_list = left_list[::-1] + sorted_list + right_list
|
||||
return all_list
|
||||
|
||||
|
||||
def point_pair2poly(point_pair_list):
|
||||
"""
|
||||
Transfer vertical point_pairs into poly point in clockwise.
|
||||
"""
|
||||
point_num = len(point_pair_list) * 2
|
||||
point_list = [0] * point_num
|
||||
for idx, point_pair in enumerate(point_pair_list):
|
||||
point_list[idx] = point_pair[0]
|
||||
point_list[point_num - 1 - idx] = point_pair[1]
|
||||
return np.array(point_list).reshape(-1, 2)
|
||||
|
||||
|
||||
def shrink_quad_along_width(quad, begin_width_ratio=0.0, end_width_ratio=1.0):
|
||||
ratio_pair = np.array([[begin_width_ratio], [end_width_ratio]], dtype=np.float32)
|
||||
p0_1 = quad[0] + (quad[1] - quad[0]) * ratio_pair
|
||||
p3_2 = quad[3] + (quad[2] - quad[3]) * ratio_pair
|
||||
return np.array([p0_1[0], p0_1[1], p3_2[1], p3_2[0]])
|
||||
|
||||
|
||||
def expand_poly_along_width(poly, shrink_ratio_of_width=0.3):
|
||||
"""
|
||||
expand poly along width.
|
||||
"""
|
||||
point_num = poly.shape[0]
|
||||
left_quad = np.array([poly[0], poly[1], poly[-2], poly[-1]], dtype=np.float32)
|
||||
left_ratio = (
|
||||
-shrink_ratio_of_width
|
||||
* np.linalg.norm(left_quad[0] - left_quad[3])
|
||||
/ (np.linalg.norm(left_quad[0] - left_quad[1]) + 1e-6)
|
||||
)
|
||||
left_quad_expand = shrink_quad_along_width(left_quad, left_ratio, 1.0)
|
||||
right_quad = np.array(
|
||||
[
|
||||
poly[point_num // 2 - 2],
|
||||
poly[point_num // 2 - 1],
|
||||
poly[point_num // 2],
|
||||
poly[point_num // 2 + 1],
|
||||
],
|
||||
dtype=np.float32,
|
||||
)
|
||||
right_ratio = 1.0 + shrink_ratio_of_width * np.linalg.norm(
|
||||
right_quad[0] - right_quad[3]
|
||||
) / (np.linalg.norm(right_quad[0] - right_quad[1]) + 1e-6)
|
||||
right_quad_expand = shrink_quad_along_width(right_quad, 0.0, right_ratio)
|
||||
poly[0] = left_quad_expand[0]
|
||||
poly[-1] = left_quad_expand[-1]
|
||||
poly[point_num // 2 - 1] = right_quad_expand[1]
|
||||
poly[point_num // 2] = right_quad_expand[2]
|
||||
return poly
|
||||
|
||||
|
||||
def restore_poly(
|
||||
instance_yxs_list, seq_strs, p_border, ratio_w, ratio_h, src_w, src_h, valid_set
|
||||
):
|
||||
poly_list = []
|
||||
keep_str_list = []
|
||||
for yx_center_line, keep_str in zip(instance_yxs_list, seq_strs):
|
||||
if len(keep_str) < 2:
|
||||
print("--> too short, {}".format(keep_str))
|
||||
continue
|
||||
|
||||
offset_expand = 1.0
|
||||
if valid_set == "totaltext":
|
||||
offset_expand = 1.2
|
||||
|
||||
point_pair_list = []
|
||||
for y, x in yx_center_line:
|
||||
offset = p_border[:, y, x].reshape(2, 2) * offset_expand
|
||||
ori_yx = np.array([y, x], dtype=np.float32)
|
||||
point_pair = (
|
||||
(ori_yx + offset)[:, ::-1]
|
||||
* 4.0
|
||||
/ np.array([ratio_w, ratio_h]).reshape(-1, 2)
|
||||
)
|
||||
point_pair_list.append(point_pair)
|
||||
|
||||
detected_poly = point_pair2poly(point_pair_list)
|
||||
detected_poly = expand_poly_along_width(
|
||||
detected_poly, shrink_ratio_of_width=0.2
|
||||
)
|
||||
detected_poly[:, 0] = np.clip(detected_poly[:, 0], a_min=0, a_max=src_w)
|
||||
detected_poly[:, 1] = np.clip(detected_poly[:, 1], a_min=0, a_max=src_h)
|
||||
|
||||
keep_str_list.append(keep_str)
|
||||
if valid_set == "partvgg":
|
||||
middle_point = len(detected_poly) // 2
|
||||
detected_poly = detected_poly[[0, middle_point - 1, middle_point, -1], :]
|
||||
poly_list.append(detected_poly)
|
||||
elif valid_set == "totaltext":
|
||||
poly_list.append(detected_poly)
|
||||
else:
|
||||
print("--> Not supported format.")
|
||||
exit(-1)
|
||||
return poly_list, keep_str_list
|
||||
|
||||
|
||||
def generate_pivot_list_fast(
|
||||
p_score,
|
||||
p_char_maps,
|
||||
f_direction,
|
||||
Lexicon_Table,
|
||||
score_thresh=0.5,
|
||||
point_gather_mode=None,
|
||||
):
|
||||
"""
|
||||
return center point and end point of TCL instance; filter with the char maps;
|
||||
"""
|
||||
p_score = p_score[0]
|
||||
f_direction = f_direction.transpose(1, 2, 0)
|
||||
p_tcl_map = (p_score > score_thresh) * 1.0
|
||||
skeleton_map = thin(p_tcl_map.astype(np.uint8))
|
||||
instance_count, instance_label_map = cv2.connectedComponents(
|
||||
skeleton_map.astype(np.uint8), connectivity=8
|
||||
)
|
||||
|
||||
# get TCL Instance
|
||||
all_pos_yxs = []
|
||||
if instance_count > 0:
|
||||
for instance_id in range(1, instance_count):
|
||||
pos_list = []
|
||||
ys, xs = np.where(instance_label_map == instance_id)
|
||||
pos_list = list(zip(ys, xs))
|
||||
|
||||
if len(pos_list) < 3:
|
||||
continue
|
||||
|
||||
pos_list_sorted = sort_and_expand_with_direction_v2(
|
||||
pos_list, f_direction, p_tcl_map
|
||||
)
|
||||
all_pos_yxs.append(pos_list_sorted)
|
||||
|
||||
p_char_maps = p_char_maps.transpose([1, 2, 0])
|
||||
decoded_str, keep_yxs_list = ctc_decoder_for_image(
|
||||
all_pos_yxs,
|
||||
logits_map=p_char_maps,
|
||||
Lexicon_Table=Lexicon_Table,
|
||||
point_gather_mode=point_gather_mode,
|
||||
)
|
||||
return keep_yxs_list, decoded_str
|
||||
|
||||
|
||||
def extract_main_direction(pos_list, f_direction):
|
||||
"""
|
||||
f_direction: h x w x 2
|
||||
pos_list: [[y, x], [y, x], [y, x] ...]
|
||||
"""
|
||||
pos_list = np.array(pos_list)
|
||||
point_direction = f_direction[pos_list[:, 0], pos_list[:, 1]]
|
||||
point_direction = point_direction[:, ::-1] # x, y -> y, x
|
||||
average_direction = np.mean(point_direction, axis=0, keepdims=True)
|
||||
average_direction = average_direction / (np.linalg.norm(average_direction) + 1e-6)
|
||||
return average_direction
|
||||
|
||||
|
||||
def sort_by_direction_with_image_id_deprecated(pos_list, f_direction):
|
||||
"""
|
||||
f_direction: h x w x 2
|
||||
pos_list: [[id, y, x], [id, y, x], [id, y, x] ...]
|
||||
"""
|
||||
pos_list_full = np.array(pos_list).reshape(-1, 3)
|
||||
pos_list = pos_list_full[:, 1:]
|
||||
point_direction = f_direction[pos_list[:, 0], pos_list[:, 1]] # x, y
|
||||
point_direction = point_direction[:, ::-1] # x, y -> y, x
|
||||
average_direction = np.mean(point_direction, axis=0, keepdims=True)
|
||||
pos_proj_leng = np.sum(pos_list * average_direction, axis=1)
|
||||
sorted_list = pos_list_full[np.argsort(pos_proj_leng)].tolist()
|
||||
return sorted_list
|
||||
|
||||
|
||||
def sort_by_direction_with_image_id(pos_list, f_direction):
|
||||
"""
|
||||
f_direction: h x w x 2
|
||||
pos_list: [[y, x], [y, x], [y, x] ...]
|
||||
"""
|
||||
|
||||
def sort_part_with_direction(pos_list_full, point_direction):
|
||||
pos_list_full = np.array(pos_list_full).reshape(-1, 3)
|
||||
pos_list = pos_list_full[:, 1:]
|
||||
point_direction = np.array(point_direction).reshape(-1, 2)
|
||||
average_direction = np.mean(point_direction, axis=0, keepdims=True)
|
||||
pos_proj_leng = np.sum(pos_list * average_direction, axis=1)
|
||||
sorted_list = pos_list_full[np.argsort(pos_proj_leng)].tolist()
|
||||
sorted_direction = point_direction[np.argsort(pos_proj_leng)].tolist()
|
||||
return sorted_list, sorted_direction
|
||||
|
||||
pos_list = np.array(pos_list).reshape(-1, 3)
|
||||
point_direction = f_direction[pos_list[:, 1], pos_list[:, 2]] # x, y
|
||||
point_direction = point_direction[:, ::-1] # x, y -> y, x
|
||||
sorted_point, sorted_direction = sort_part_with_direction(pos_list, point_direction)
|
||||
|
||||
point_num = len(sorted_point)
|
||||
if point_num >= 16:
|
||||
middle_num = point_num // 2
|
||||
first_part_point = sorted_point[:middle_num]
|
||||
first_point_direction = sorted_direction[:middle_num]
|
||||
sorted_fist_part_point, sorted_fist_part_direction = sort_part_with_direction(
|
||||
first_part_point, first_point_direction
|
||||
)
|
||||
|
||||
last_part_point = sorted_point[middle_num:]
|
||||
last_point_direction = sorted_direction[middle_num:]
|
||||
sorted_last_part_point, sorted_last_part_direction = sort_part_with_direction(
|
||||
last_part_point, last_point_direction
|
||||
)
|
||||
sorted_point = sorted_fist_part_point + sorted_last_part_point
|
||||
sorted_direction = sorted_fist_part_direction + sorted_last_part_direction
|
||||
|
||||
return sorted_point
|
||||
624
ppocr/utils/e2e_utils/extract_textpoint_slow.py
Normal file
624
ppocr/utils/e2e_utils/extract_textpoint_slow.py
Normal file
@@ -0,0 +1,624 @@
|
||||
# Copyright (c) 2021 PaddlePaddle Authors. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""Contains various CTC decoders."""
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import cv2
|
||||
import math
|
||||
|
||||
import numpy as np
|
||||
from itertools import groupby
|
||||
from skimage.morphology._skeletonize import thin
|
||||
|
||||
|
||||
def get_dict(character_dict_path):
|
||||
character_str = ""
|
||||
with open(character_dict_path, "rb") as fin:
|
||||
lines = fin.readlines()
|
||||
for line in lines:
|
||||
line = line.decode("utf-8").strip("\n").strip("\r\n")
|
||||
character_str += line
|
||||
dict_character = list(character_str)
|
||||
return dict_character
|
||||
|
||||
|
||||
def point_pair2poly(point_pair_list):
|
||||
"""
|
||||
Transfer vertical point_pairs into poly point in clockwise.
|
||||
"""
|
||||
pair_length_list = []
|
||||
for point_pair in point_pair_list:
|
||||
pair_length = np.linalg.norm(point_pair[0] - point_pair[1])
|
||||
pair_length_list.append(pair_length)
|
||||
pair_length_list = np.array(pair_length_list)
|
||||
pair_info = (
|
||||
pair_length_list.max(),
|
||||
pair_length_list.min(),
|
||||
pair_length_list.mean(),
|
||||
)
|
||||
|
||||
point_num = len(point_pair_list) * 2
|
||||
point_list = [0] * point_num
|
||||
for idx, point_pair in enumerate(point_pair_list):
|
||||
point_list[idx] = point_pair[0]
|
||||
point_list[point_num - 1 - idx] = point_pair[1]
|
||||
return np.array(point_list).reshape(-1, 2), pair_info
|
||||
|
||||
|
||||
def shrink_quad_along_width(quad, begin_width_ratio=0.0, end_width_ratio=1.0):
|
||||
"""
|
||||
Generate shrink_quad_along_width.
|
||||
"""
|
||||
ratio_pair = np.array([[begin_width_ratio], [end_width_ratio]], dtype=np.float32)
|
||||
p0_1 = quad[0] + (quad[1] - quad[0]) * ratio_pair
|
||||
p3_2 = quad[3] + (quad[2] - quad[3]) * ratio_pair
|
||||
return np.array([p0_1[0], p0_1[1], p3_2[1], p3_2[0]])
|
||||
|
||||
|
||||
def expand_poly_along_width(poly, shrink_ratio_of_width=0.3):
|
||||
"""
|
||||
expand poly along width.
|
||||
"""
|
||||
point_num = poly.shape[0]
|
||||
left_quad = np.array([poly[0], poly[1], poly[-2], poly[-1]], dtype=np.float32)
|
||||
left_ratio = (
|
||||
-shrink_ratio_of_width
|
||||
* np.linalg.norm(left_quad[0] - left_quad[3])
|
||||
/ (np.linalg.norm(left_quad[0] - left_quad[1]) + 1e-6)
|
||||
)
|
||||
left_quad_expand = shrink_quad_along_width(left_quad, left_ratio, 1.0)
|
||||
right_quad = np.array(
|
||||
[
|
||||
poly[point_num // 2 - 2],
|
||||
poly[point_num // 2 - 1],
|
||||
poly[point_num // 2],
|
||||
poly[point_num // 2 + 1],
|
||||
],
|
||||
dtype=np.float32,
|
||||
)
|
||||
right_ratio = 1.0 + shrink_ratio_of_width * np.linalg.norm(
|
||||
right_quad[0] - right_quad[3]
|
||||
) / (np.linalg.norm(right_quad[0] - right_quad[1]) + 1e-6)
|
||||
right_quad_expand = shrink_quad_along_width(right_quad, 0.0, right_ratio)
|
||||
poly[0] = left_quad_expand[0]
|
||||
poly[-1] = left_quad_expand[-1]
|
||||
poly[point_num // 2 - 1] = right_quad_expand[1]
|
||||
poly[point_num // 2] = right_quad_expand[2]
|
||||
return poly
|
||||
|
||||
|
||||
def softmax(logits):
|
||||
"""
|
||||
logits: N x d
|
||||
"""
|
||||
max_value = np.max(logits, axis=1, keepdims=True)
|
||||
exp = np.exp(logits - max_value)
|
||||
exp_sum = np.sum(exp, axis=1, keepdims=True)
|
||||
dist = exp / exp_sum
|
||||
return dist
|
||||
|
||||
|
||||
def get_keep_pos_idxs(labels, remove_blank=None):
|
||||
"""
|
||||
Remove duplicate and get pos idxs of keep items.
|
||||
The value of keep_blank should be [None, 95].
|
||||
"""
|
||||
duplicate_len_list = []
|
||||
keep_pos_idx_list = []
|
||||
keep_char_idx_list = []
|
||||
for k, v_ in groupby(labels):
|
||||
current_len = len(list(v_))
|
||||
if k != remove_blank:
|
||||
current_idx = int(sum(duplicate_len_list) + current_len // 2)
|
||||
keep_pos_idx_list.append(current_idx)
|
||||
keep_char_idx_list.append(k)
|
||||
duplicate_len_list.append(current_len)
|
||||
return keep_char_idx_list, keep_pos_idx_list
|
||||
|
||||
|
||||
def remove_blank(labels, blank=0):
|
||||
new_labels = [x for x in labels if x != blank]
|
||||
return new_labels
|
||||
|
||||
|
||||
def insert_blank(labels, blank=0):
|
||||
new_labels = [blank]
|
||||
for l in labels:
|
||||
new_labels += [l, blank]
|
||||
return new_labels
|
||||
|
||||
|
||||
def ctc_greedy_decoder(probs_seq, blank=95, keep_blank_in_idxs=True):
|
||||
"""
|
||||
CTC greedy (best path) decoder.
|
||||
"""
|
||||
raw_str = np.argmax(np.array(probs_seq), axis=1)
|
||||
remove_blank_in_pos = None if keep_blank_in_idxs else blank
|
||||
dedup_str, keep_idx_list = get_keep_pos_idxs(
|
||||
raw_str, remove_blank=remove_blank_in_pos
|
||||
)
|
||||
dst_str = remove_blank(dedup_str, blank=blank)
|
||||
return dst_str, keep_idx_list
|
||||
|
||||
|
||||
def instance_ctc_greedy_decoder(gather_info, logits_map, keep_blank_in_idxs=True):
|
||||
"""
|
||||
gather_info: [[x, y], [x, y] ...]
|
||||
logits_map: H x W X (n_chars + 1)
|
||||
"""
|
||||
_, _, C = logits_map.shape
|
||||
ys, xs = zip(*gather_info)
|
||||
logits_seq = logits_map[list(ys), list(xs)] # n x 96
|
||||
probs_seq = softmax(logits_seq)
|
||||
dst_str, keep_idx_list = ctc_greedy_decoder(
|
||||
probs_seq, blank=C - 1, keep_blank_in_idxs=keep_blank_in_idxs
|
||||
)
|
||||
keep_gather_list = [gather_info[idx] for idx in keep_idx_list]
|
||||
return dst_str, keep_gather_list
|
||||
|
||||
|
||||
def ctc_decoder_for_image(gather_info_list, logits_map, keep_blank_in_idxs=True):
|
||||
"""
|
||||
CTC decoder using multiple processes.
|
||||
"""
|
||||
decoder_results = []
|
||||
for gather_info in gather_info_list:
|
||||
res = instance_ctc_greedy_decoder(
|
||||
gather_info, logits_map, keep_blank_in_idxs=keep_blank_in_idxs
|
||||
)
|
||||
decoder_results.append(res)
|
||||
return decoder_results
|
||||
|
||||
|
||||
def sort_with_direction(pos_list, f_direction):
|
||||
"""
|
||||
f_direction: h x w x 2
|
||||
pos_list: [[y, x], [y, x], [y, x] ...]
|
||||
"""
|
||||
|
||||
def sort_part_with_direction(pos_list, point_direction):
|
||||
pos_list = np.array(pos_list).reshape(-1, 2)
|
||||
point_direction = np.array(point_direction).reshape(-1, 2)
|
||||
average_direction = np.mean(point_direction, axis=0, keepdims=True)
|
||||
pos_proj_leng = np.sum(pos_list * average_direction, axis=1)
|
||||
sorted_list = pos_list[np.argsort(pos_proj_leng)].tolist()
|
||||
sorted_direction = point_direction[np.argsort(pos_proj_leng)].tolist()
|
||||
return sorted_list, sorted_direction
|
||||
|
||||
pos_list = np.array(pos_list).reshape(-1, 2)
|
||||
point_direction = f_direction[pos_list[:, 0], pos_list[:, 1]] # x, y
|
||||
point_direction = point_direction[:, ::-1] # x, y -> y, x
|
||||
sorted_point, sorted_direction = sort_part_with_direction(pos_list, point_direction)
|
||||
|
||||
point_num = len(sorted_point)
|
||||
if point_num >= 16:
|
||||
middle_num = point_num // 2
|
||||
first_part_point = sorted_point[:middle_num]
|
||||
first_point_direction = sorted_direction[:middle_num]
|
||||
sorted_fist_part_point, sorted_fist_part_direction = sort_part_with_direction(
|
||||
first_part_point, first_point_direction
|
||||
)
|
||||
|
||||
last_part_point = sorted_point[middle_num:]
|
||||
last_point_direction = sorted_direction[middle_num:]
|
||||
sorted_last_part_point, sorted_last_part_direction = sort_part_with_direction(
|
||||
last_part_point, last_point_direction
|
||||
)
|
||||
sorted_point = sorted_fist_part_point + sorted_last_part_point
|
||||
sorted_direction = sorted_fist_part_direction + sorted_last_part_direction
|
||||
|
||||
return sorted_point, np.array(sorted_direction)
|
||||
|
||||
|
||||
def add_id(pos_list, image_id=0):
|
||||
"""
|
||||
Add id for gather feature, for inference.
|
||||
"""
|
||||
new_list = []
|
||||
for item in pos_list:
|
||||
new_list.append((image_id, item[0], item[1]))
|
||||
return new_list
|
||||
|
||||
|
||||
def sort_and_expand_with_direction(pos_list, f_direction):
|
||||
"""
|
||||
f_direction: h x w x 2
|
||||
pos_list: [[y, x], [y, x], [y, x] ...]
|
||||
"""
|
||||
h, w, _ = f_direction.shape
|
||||
sorted_list, point_direction = sort_with_direction(pos_list, f_direction)
|
||||
|
||||
# expand along
|
||||
point_num = len(sorted_list)
|
||||
sub_direction_len = max(point_num // 3, 2)
|
||||
left_direction = point_direction[:sub_direction_len, :]
|
||||
right_dirction = point_direction[point_num - sub_direction_len :, :]
|
||||
|
||||
left_average_direction = -np.mean(left_direction, axis=0, keepdims=True)
|
||||
left_average_len = np.linalg.norm(left_average_direction)
|
||||
left_start = np.array(sorted_list[0])
|
||||
left_step = left_average_direction / (left_average_len + 1e-6)
|
||||
|
||||
right_average_direction = np.mean(right_dirction, axis=0, keepdims=True)
|
||||
right_average_len = np.linalg.norm(right_average_direction)
|
||||
right_step = right_average_direction / (right_average_len + 1e-6)
|
||||
right_start = np.array(sorted_list[-1])
|
||||
|
||||
append_num = max(int((left_average_len + right_average_len) / 2.0 * 0.15), 1)
|
||||
left_list = []
|
||||
right_list = []
|
||||
for i in range(append_num):
|
||||
ly, lx = (
|
||||
np.round(left_start + left_step * (i + 1))
|
||||
.flatten()
|
||||
.astype("int32")
|
||||
.tolist()
|
||||
)
|
||||
if ly < h and lx < w and (ly, lx) not in left_list:
|
||||
left_list.append((ly, lx))
|
||||
ry, rx = (
|
||||
np.round(right_start + right_step * (i + 1))
|
||||
.flatten()
|
||||
.astype("int32")
|
||||
.tolist()
|
||||
)
|
||||
if ry < h and rx < w and (ry, rx) not in right_list:
|
||||
right_list.append((ry, rx))
|
||||
|
||||
all_list = left_list[::-1] + sorted_list + right_list
|
||||
return all_list
|
||||
|
||||
|
||||
def sort_and_expand_with_direction_v2(pos_list, f_direction, binary_tcl_map):
|
||||
"""
|
||||
f_direction: h x w x 2
|
||||
pos_list: [[y, x], [y, x], [y, x] ...]
|
||||
binary_tcl_map: h x w
|
||||
"""
|
||||
h, w, _ = f_direction.shape
|
||||
sorted_list, point_direction = sort_with_direction(pos_list, f_direction)
|
||||
|
||||
# expand along
|
||||
point_num = len(sorted_list)
|
||||
sub_direction_len = max(point_num // 3, 2)
|
||||
left_direction = point_direction[:sub_direction_len, :]
|
||||
right_dirction = point_direction[point_num - sub_direction_len :, :]
|
||||
|
||||
left_average_direction = -np.mean(left_direction, axis=0, keepdims=True)
|
||||
left_average_len = np.linalg.norm(left_average_direction)
|
||||
left_start = np.array(sorted_list[0])
|
||||
left_step = left_average_direction / (left_average_len + 1e-6)
|
||||
|
||||
right_average_direction = np.mean(right_dirction, axis=0, keepdims=True)
|
||||
right_average_len = np.linalg.norm(right_average_direction)
|
||||
right_step = right_average_direction / (right_average_len + 1e-6)
|
||||
right_start = np.array(sorted_list[-1])
|
||||
|
||||
append_num = max(int((left_average_len + right_average_len) / 2.0 * 0.15), 1)
|
||||
max_append_num = 2 * append_num
|
||||
|
||||
left_list = []
|
||||
right_list = []
|
||||
for i in range(max_append_num):
|
||||
ly, lx = (
|
||||
np.round(left_start + left_step * (i + 1))
|
||||
.flatten()
|
||||
.astype("int32")
|
||||
.tolist()
|
||||
)
|
||||
if ly < h and lx < w and (ly, lx) not in left_list:
|
||||
if binary_tcl_map[ly, lx] > 0.5:
|
||||
left_list.append((ly, lx))
|
||||
else:
|
||||
break
|
||||
|
||||
for i in range(max_append_num):
|
||||
ry, rx = (
|
||||
np.round(right_start + right_step * (i + 1))
|
||||
.flatten()
|
||||
.astype("int32")
|
||||
.tolist()
|
||||
)
|
||||
if ry < h and rx < w and (ry, rx) not in right_list:
|
||||
if binary_tcl_map[ry, rx] > 0.5:
|
||||
right_list.append((ry, rx))
|
||||
else:
|
||||
break
|
||||
|
||||
all_list = left_list[::-1] + sorted_list + right_list
|
||||
return all_list
|
||||
|
||||
|
||||
def generate_pivot_list_curved(
|
||||
p_score,
|
||||
p_char_maps,
|
||||
f_direction,
|
||||
score_thresh=0.5,
|
||||
is_expand=True,
|
||||
is_backbone=False,
|
||||
image_id=0,
|
||||
):
|
||||
"""
|
||||
return center point and end point of TCL instance; filter with the char maps;
|
||||
"""
|
||||
p_score = p_score[0]
|
||||
f_direction = f_direction.transpose(1, 2, 0)
|
||||
p_tcl_map = (p_score > score_thresh) * 1.0
|
||||
skeleton_map = thin(p_tcl_map)
|
||||
instance_count, instance_label_map = cv2.connectedComponents(
|
||||
skeleton_map.astype(np.uint8), connectivity=8
|
||||
)
|
||||
|
||||
# get TCL Instance
|
||||
all_pos_yxs = []
|
||||
center_pos_yxs = []
|
||||
end_points_yxs = []
|
||||
instance_center_pos_yxs = []
|
||||
pred_strs = []
|
||||
if instance_count > 0:
|
||||
for instance_id in range(1, instance_count):
|
||||
pos_list = []
|
||||
ys, xs = np.where(instance_label_map == instance_id)
|
||||
pos_list = list(zip(ys, xs))
|
||||
|
||||
### FIX-ME, eliminate outlier
|
||||
if len(pos_list) < 3:
|
||||
continue
|
||||
|
||||
if is_expand:
|
||||
pos_list_sorted = sort_and_expand_with_direction_v2(
|
||||
pos_list, f_direction, p_tcl_map
|
||||
)
|
||||
else:
|
||||
pos_list_sorted, _ = sort_with_direction(pos_list, f_direction)
|
||||
all_pos_yxs.append(pos_list_sorted)
|
||||
|
||||
# use decoder to filter background points.
|
||||
p_char_maps = p_char_maps.transpose([1, 2, 0])
|
||||
decode_res = ctc_decoder_for_image(
|
||||
all_pos_yxs, logits_map=p_char_maps, keep_blank_in_idxs=True
|
||||
)
|
||||
for decoded_str, keep_yxs_list in decode_res:
|
||||
if is_backbone:
|
||||
keep_yxs_list_with_id = add_id(keep_yxs_list, image_id=image_id)
|
||||
instance_center_pos_yxs.append(keep_yxs_list_with_id)
|
||||
pred_strs.append(decoded_str)
|
||||
else:
|
||||
end_points_yxs.extend((keep_yxs_list[0], keep_yxs_list[-1]))
|
||||
center_pos_yxs.extend(keep_yxs_list)
|
||||
|
||||
if is_backbone:
|
||||
return pred_strs, instance_center_pos_yxs
|
||||
else:
|
||||
return center_pos_yxs, end_points_yxs
|
||||
|
||||
|
||||
def generate_pivot_list_horizontal(
|
||||
p_score, p_char_maps, f_direction, score_thresh=0.5, is_backbone=False, image_id=0
|
||||
):
|
||||
"""
|
||||
return center point and end point of TCL instance; filter with the char maps;
|
||||
"""
|
||||
p_score = p_score[0]
|
||||
f_direction = f_direction.transpose(1, 2, 0)
|
||||
p_tcl_map_bi = (p_score > score_thresh) * 1.0
|
||||
instance_count, instance_label_map = cv2.connectedComponents(
|
||||
p_tcl_map_bi.astype(np.uint8), connectivity=8
|
||||
)
|
||||
|
||||
# get TCL Instance
|
||||
all_pos_yxs = []
|
||||
center_pos_yxs = []
|
||||
end_points_yxs = []
|
||||
instance_center_pos_yxs = []
|
||||
|
||||
if instance_count > 0:
|
||||
for instance_id in range(1, instance_count):
|
||||
pos_list = []
|
||||
ys, xs = np.where(instance_label_map == instance_id)
|
||||
pos_list = list(zip(ys, xs))
|
||||
|
||||
### FIX-ME, eliminate outlier
|
||||
if len(pos_list) < 5:
|
||||
continue
|
||||
|
||||
# add rule here
|
||||
main_direction = extract_main_direction(pos_list, f_direction) # y x
|
||||
reference_directin = np.array([0, 1]).reshape([-1, 2]) # y x
|
||||
is_h_angle = abs(np.sum(main_direction * reference_directin)) < math.cos(
|
||||
math.pi / 180 * 70
|
||||
)
|
||||
|
||||
point_yxs = np.array(pos_list)
|
||||
max_y, max_x = np.max(point_yxs, axis=0)
|
||||
min_y, min_x = np.min(point_yxs, axis=0)
|
||||
is_h_len = (max_y - min_y) < 1.5 * (max_x - min_x)
|
||||
|
||||
pos_list_final = []
|
||||
if is_h_len:
|
||||
xs = np.unique(xs)
|
||||
for x in xs:
|
||||
ys = instance_label_map[:, x].copy().reshape((-1,))
|
||||
y = int(np.where(ys == instance_id)[0].mean())
|
||||
pos_list_final.append((y, x))
|
||||
else:
|
||||
ys = np.unique(ys)
|
||||
for y in ys:
|
||||
xs = instance_label_map[y, :].copy().reshape((-1,))
|
||||
x = int(np.where(xs == instance_id)[0].mean())
|
||||
pos_list_final.append((y, x))
|
||||
|
||||
pos_list_sorted, _ = sort_with_direction(pos_list_final, f_direction)
|
||||
all_pos_yxs.append(pos_list_sorted)
|
||||
|
||||
# use decoder to filter background points.
|
||||
p_char_maps = p_char_maps.transpose([1, 2, 0])
|
||||
decode_res = ctc_decoder_for_image(
|
||||
all_pos_yxs, logits_map=p_char_maps, keep_blank_in_idxs=True
|
||||
)
|
||||
for decoded_str, keep_yxs_list in decode_res:
|
||||
if is_backbone:
|
||||
keep_yxs_list_with_id = add_id(keep_yxs_list, image_id=image_id)
|
||||
instance_center_pos_yxs.append(keep_yxs_list_with_id)
|
||||
else:
|
||||
end_points_yxs.extend((keep_yxs_list[0], keep_yxs_list[-1]))
|
||||
center_pos_yxs.extend(keep_yxs_list)
|
||||
|
||||
if is_backbone:
|
||||
return instance_center_pos_yxs
|
||||
else:
|
||||
return center_pos_yxs, end_points_yxs
|
||||
|
||||
|
||||
def generate_pivot_list_slow(
|
||||
p_score,
|
||||
p_char_maps,
|
||||
f_direction,
|
||||
score_thresh=0.5,
|
||||
is_backbone=False,
|
||||
is_curved=True,
|
||||
image_id=0,
|
||||
):
|
||||
"""
|
||||
Warp all the function together.
|
||||
"""
|
||||
if is_curved:
|
||||
return generate_pivot_list_curved(
|
||||
p_score,
|
||||
p_char_maps,
|
||||
f_direction,
|
||||
score_thresh=score_thresh,
|
||||
is_expand=True,
|
||||
is_backbone=is_backbone,
|
||||
image_id=image_id,
|
||||
)
|
||||
else:
|
||||
return generate_pivot_list_horizontal(
|
||||
p_score,
|
||||
p_char_maps,
|
||||
f_direction,
|
||||
score_thresh=score_thresh,
|
||||
is_backbone=is_backbone,
|
||||
image_id=image_id,
|
||||
)
|
||||
|
||||
|
||||
# for refine module
|
||||
def extract_main_direction(pos_list, f_direction):
|
||||
"""
|
||||
f_direction: h x w x 2
|
||||
pos_list: [[y, x], [y, x], [y, x] ...]
|
||||
"""
|
||||
pos_list = np.array(pos_list)
|
||||
point_direction = f_direction[pos_list[:, 0], pos_list[:, 1]]
|
||||
point_direction = point_direction[:, ::-1] # x, y -> y, x
|
||||
average_direction = np.mean(point_direction, axis=0, keepdims=True)
|
||||
average_direction = average_direction / (np.linalg.norm(average_direction) + 1e-6)
|
||||
return average_direction
|
||||
|
||||
|
||||
def sort_by_direction_with_image_id_deprecated(pos_list, f_direction):
|
||||
"""
|
||||
f_direction: h x w x 2
|
||||
pos_list: [[id, y, x], [id, y, x], [id, y, x] ...]
|
||||
"""
|
||||
pos_list_full = np.array(pos_list).reshape(-1, 3)
|
||||
pos_list = pos_list_full[:, 1:]
|
||||
point_direction = f_direction[pos_list[:, 0], pos_list[:, 1]] # x, y
|
||||
point_direction = point_direction[:, ::-1] # x, y -> y, x
|
||||
average_direction = np.mean(point_direction, axis=0, keepdims=True)
|
||||
pos_proj_leng = np.sum(pos_list * average_direction, axis=1)
|
||||
sorted_list = pos_list_full[np.argsort(pos_proj_leng)].tolist()
|
||||
return sorted_list
|
||||
|
||||
|
||||
def sort_by_direction_with_image_id(pos_list, f_direction):
|
||||
"""
|
||||
f_direction: h x w x 2
|
||||
pos_list: [[y, x], [y, x], [y, x] ...]
|
||||
"""
|
||||
|
||||
def sort_part_with_direction(pos_list_full, point_direction):
|
||||
pos_list_full = np.array(pos_list_full).reshape(-1, 3)
|
||||
pos_list = pos_list_full[:, 1:]
|
||||
point_direction = np.array(point_direction).reshape(-1, 2)
|
||||
average_direction = np.mean(point_direction, axis=0, keepdims=True)
|
||||
pos_proj_leng = np.sum(pos_list * average_direction, axis=1)
|
||||
sorted_list = pos_list_full[np.argsort(pos_proj_leng)].tolist()
|
||||
sorted_direction = point_direction[np.argsort(pos_proj_leng)].tolist()
|
||||
return sorted_list, sorted_direction
|
||||
|
||||
pos_list = np.array(pos_list).reshape(-1, 3)
|
||||
point_direction = f_direction[pos_list[:, 1], pos_list[:, 2]] # x, y
|
||||
point_direction = point_direction[:, ::-1] # x, y -> y, x
|
||||
sorted_point, sorted_direction = sort_part_with_direction(pos_list, point_direction)
|
||||
|
||||
point_num = len(sorted_point)
|
||||
if point_num >= 16:
|
||||
middle_num = point_num // 2
|
||||
first_part_point = sorted_point[:middle_num]
|
||||
first_point_direction = sorted_direction[:middle_num]
|
||||
sorted_fist_part_point, sorted_fist_part_direction = sort_part_with_direction(
|
||||
first_part_point, first_point_direction
|
||||
)
|
||||
|
||||
last_part_point = sorted_point[middle_num:]
|
||||
last_point_direction = sorted_direction[middle_num:]
|
||||
sorted_last_part_point, sorted_last_part_direction = sort_part_with_direction(
|
||||
last_part_point, last_point_direction
|
||||
)
|
||||
sorted_point = sorted_fist_part_point + sorted_last_part_point
|
||||
sorted_direction = sorted_fist_part_direction + sorted_last_part_direction
|
||||
|
||||
return sorted_point
|
||||
|
||||
|
||||
def generate_pivot_list_tt_inference(
|
||||
p_score,
|
||||
p_char_maps,
|
||||
f_direction,
|
||||
score_thresh=0.5,
|
||||
is_backbone=False,
|
||||
is_curved=True,
|
||||
image_id=0,
|
||||
):
|
||||
"""
|
||||
return center point and end point of TCL instance; filter with the char maps;
|
||||
"""
|
||||
p_score = p_score[0]
|
||||
f_direction = f_direction.transpose(1, 2, 0)
|
||||
p_tcl_map = (p_score > score_thresh) * 1.0
|
||||
skeleton_map = thin(p_tcl_map)
|
||||
instance_count, instance_label_map = cv2.connectedComponents(
|
||||
skeleton_map.astype(np.uint8), connectivity=8
|
||||
)
|
||||
|
||||
# get TCL Instance
|
||||
all_pos_yxs = []
|
||||
if instance_count > 0:
|
||||
for instance_id in range(1, instance_count):
|
||||
pos_list = []
|
||||
ys, xs = np.where(instance_label_map == instance_id)
|
||||
pos_list = list(zip(ys, xs))
|
||||
### FIX-ME, eliminate outlier
|
||||
if len(pos_list) < 3:
|
||||
continue
|
||||
pos_list_sorted = sort_and_expand_with_direction_v2(
|
||||
pos_list, f_direction, p_tcl_map
|
||||
)
|
||||
pos_list_sorted_with_id = add_id(pos_list_sorted, image_id=image_id)
|
||||
all_pos_yxs.append(pos_list_sorted_with_id)
|
||||
return all_pos_yxs
|
||||
179
ppocr/utils/e2e_utils/pgnet_pp_utils.py
Normal file
179
ppocr/utils/e2e_utils/pgnet_pp_utils.py
Normal file
@@ -0,0 +1,179 @@
|
||||
# Copyright (c) 2021 PaddlePaddle Authors. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
import paddle
|
||||
import os
|
||||
import sys
|
||||
|
||||
__dir__ = os.path.dirname(__file__)
|
||||
sys.path.append(__dir__)
|
||||
sys.path.append(os.path.join(__dir__, ".."))
|
||||
from extract_textpoint_slow import *
|
||||
from extract_textpoint_fast import generate_pivot_list_fast, restore_poly
|
||||
|
||||
|
||||
class PGNet_PostProcess(object):
|
||||
# two different post-process
|
||||
def __init__(
|
||||
self,
|
||||
character_dict_path,
|
||||
valid_set,
|
||||
score_thresh,
|
||||
outs_dict,
|
||||
shape_list,
|
||||
point_gather_mode=None,
|
||||
):
|
||||
self.Lexicon_Table = get_dict(character_dict_path)
|
||||
self.valid_set = valid_set
|
||||
self.score_thresh = score_thresh
|
||||
self.outs_dict = outs_dict
|
||||
self.shape_list = shape_list
|
||||
self.point_gather_mode = point_gather_mode
|
||||
|
||||
def pg_postprocess_fast(self):
|
||||
p_score = self.outs_dict["f_score"]
|
||||
p_border = self.outs_dict["f_border"]
|
||||
p_char = self.outs_dict["f_char"]
|
||||
p_direction = self.outs_dict["f_direction"]
|
||||
if isinstance(p_score, paddle.Tensor):
|
||||
p_score = p_score[0].numpy()
|
||||
p_border = p_border[0].numpy()
|
||||
p_direction = p_direction[0].numpy()
|
||||
p_char = p_char[0].numpy()
|
||||
else:
|
||||
p_score = p_score[0]
|
||||
p_border = p_border[0]
|
||||
p_direction = p_direction[0]
|
||||
p_char = p_char[0]
|
||||
|
||||
src_h, src_w, ratio_h, ratio_w = self.shape_list[0]
|
||||
instance_yxs_list, seq_strs = generate_pivot_list_fast(
|
||||
p_score,
|
||||
p_char,
|
||||
p_direction,
|
||||
self.Lexicon_Table,
|
||||
score_thresh=self.score_thresh,
|
||||
point_gather_mode=self.point_gather_mode,
|
||||
)
|
||||
poly_list, keep_str_list = restore_poly(
|
||||
instance_yxs_list,
|
||||
seq_strs,
|
||||
p_border,
|
||||
ratio_w,
|
||||
ratio_h,
|
||||
src_w,
|
||||
src_h,
|
||||
self.valid_set,
|
||||
)
|
||||
data = {
|
||||
"points": poly_list,
|
||||
"texts": keep_str_list,
|
||||
}
|
||||
return data
|
||||
|
||||
def pg_postprocess_slow(self):
|
||||
p_score = self.outs_dict["f_score"]
|
||||
p_border = self.outs_dict["f_border"]
|
||||
p_char = self.outs_dict["f_char"]
|
||||
p_direction = self.outs_dict["f_direction"]
|
||||
if isinstance(p_score, paddle.Tensor):
|
||||
p_score = p_score[0].numpy()
|
||||
p_border = p_border[0].numpy()
|
||||
p_direction = p_direction[0].numpy()
|
||||
p_char = p_char[0].numpy()
|
||||
else:
|
||||
p_score = p_score[0]
|
||||
p_border = p_border[0]
|
||||
p_direction = p_direction[0]
|
||||
p_char = p_char[0]
|
||||
src_h, src_w, ratio_h, ratio_w = self.shape_list[0]
|
||||
is_curved = self.valid_set == "totaltext"
|
||||
char_seq_idx_set, instance_yxs_list = generate_pivot_list_slow(
|
||||
p_score,
|
||||
p_char,
|
||||
p_direction,
|
||||
score_thresh=self.score_thresh,
|
||||
is_backbone=True,
|
||||
is_curved=is_curved,
|
||||
)
|
||||
seq_strs = []
|
||||
for char_idx_set in char_seq_idx_set:
|
||||
pr_str = "".join([self.Lexicon_Table[pos] for pos in char_idx_set])
|
||||
seq_strs.append(pr_str)
|
||||
poly_list = []
|
||||
keep_str_list = []
|
||||
all_point_list = []
|
||||
all_point_pair_list = []
|
||||
for yx_center_line, keep_str in zip(instance_yxs_list, seq_strs):
|
||||
if len(yx_center_line) == 1:
|
||||
yx_center_line.append(yx_center_line[-1])
|
||||
|
||||
offset_expand = 1.0
|
||||
if self.valid_set == "totaltext":
|
||||
offset_expand = 1.2
|
||||
|
||||
point_pair_list = []
|
||||
for batch_id, y, x in yx_center_line:
|
||||
offset = p_border[:, y, x].reshape(2, 2)
|
||||
if offset_expand != 1.0:
|
||||
offset_length = np.linalg.norm(offset, axis=1, keepdims=True)
|
||||
expand_length = np.clip(
|
||||
offset_length * (offset_expand - 1), a_min=0.5, a_max=3.0
|
||||
)
|
||||
offset_detal = offset / offset_length * expand_length
|
||||
offset = offset + offset_detal
|
||||
ori_yx = np.array([y, x], dtype=np.float32)
|
||||
point_pair = (
|
||||
(ori_yx + offset)[:, ::-1]
|
||||
* 4.0
|
||||
/ np.array([ratio_w, ratio_h]).reshape(-1, 2)
|
||||
)
|
||||
point_pair_list.append(point_pair)
|
||||
|
||||
all_point_list.append(
|
||||
[int(round(x * 4.0 / ratio_w)), int(round(y * 4.0 / ratio_h))]
|
||||
)
|
||||
all_point_pair_list.append(point_pair.round().astype(np.int32).tolist())
|
||||
|
||||
detected_poly, pair_length_info = point_pair2poly(point_pair_list)
|
||||
detected_poly = expand_poly_along_width(
|
||||
detected_poly, shrink_ratio_of_width=0.2
|
||||
)
|
||||
detected_poly[:, 0] = np.clip(detected_poly[:, 0], a_min=0, a_max=src_w)
|
||||
detected_poly[:, 1] = np.clip(detected_poly[:, 1], a_min=0, a_max=src_h)
|
||||
|
||||
if len(keep_str) < 2:
|
||||
continue
|
||||
|
||||
keep_str_list.append(keep_str)
|
||||
detected_poly = np.round(detected_poly).astype("int32")
|
||||
if self.valid_set == "partvgg":
|
||||
middle_point = len(detected_poly) // 2
|
||||
detected_poly = detected_poly[
|
||||
[0, middle_point - 1, middle_point, -1], :
|
||||
]
|
||||
poly_list.append(detected_poly)
|
||||
elif self.valid_set == "totaltext":
|
||||
poly_list.append(detected_poly)
|
||||
else:
|
||||
print("--> Not supported format.")
|
||||
exit(-1)
|
||||
data = {
|
||||
"points": poly_list,
|
||||
"texts": keep_str_list,
|
||||
}
|
||||
return data
|
||||
167
ppocr/utils/e2e_utils/visual.py
Normal file
167
ppocr/utils/e2e_utils/visual.py
Normal file
@@ -0,0 +1,167 @@
|
||||
# Copyright (c) 2021 PaddlePaddle Authors. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
import numpy as np
|
||||
import cv2
|
||||
import time
|
||||
|
||||
|
||||
def resize_image(im, max_side_len=512):
|
||||
"""
|
||||
resize image to a size multiple of max_stride which is required by the network
|
||||
:param im: the resized image
|
||||
:param max_side_len: limit of max image size to avoid out of memory in gpu
|
||||
:return: the resized image and the resize ratio
|
||||
"""
|
||||
h, w, _ = im.shape
|
||||
|
||||
resize_w = w
|
||||
resize_h = h
|
||||
|
||||
if resize_h > resize_w:
|
||||
ratio = float(max_side_len) / resize_h
|
||||
else:
|
||||
ratio = float(max_side_len) / resize_w
|
||||
|
||||
resize_h = int(resize_h * ratio)
|
||||
resize_w = int(resize_w * ratio)
|
||||
|
||||
max_stride = 128
|
||||
resize_h = (resize_h + max_stride - 1) // max_stride * max_stride
|
||||
resize_w = (resize_w + max_stride - 1) // max_stride * max_stride
|
||||
im = cv2.resize(im, (int(resize_w), int(resize_h)))
|
||||
ratio_h = resize_h / float(h)
|
||||
ratio_w = resize_w / float(w)
|
||||
|
||||
return im, (ratio_h, ratio_w)
|
||||
|
||||
|
||||
def resize_image_min(im, max_side_len=512):
|
||||
""" """
|
||||
h, w, _ = im.shape
|
||||
|
||||
resize_w = w
|
||||
resize_h = h
|
||||
|
||||
if resize_h < resize_w:
|
||||
ratio = float(max_side_len) / resize_h
|
||||
else:
|
||||
ratio = float(max_side_len) / resize_w
|
||||
|
||||
resize_h = int(resize_h * ratio)
|
||||
resize_w = int(resize_w * ratio)
|
||||
|
||||
max_stride = 128
|
||||
resize_h = (resize_h + max_stride - 1) // max_stride * max_stride
|
||||
resize_w = (resize_w + max_stride - 1) // max_stride * max_stride
|
||||
im = cv2.resize(im, (int(resize_w), int(resize_h)))
|
||||
ratio_h = resize_h / float(h)
|
||||
ratio_w = resize_w / float(w)
|
||||
return im, (ratio_h, ratio_w)
|
||||
|
||||
|
||||
def resize_image_for_totaltext(im, max_side_len=512):
|
||||
""" """
|
||||
h, w, _ = im.shape
|
||||
|
||||
resize_w = w
|
||||
resize_h = h
|
||||
ratio = 1.25
|
||||
if h * ratio > max_side_len:
|
||||
ratio = float(max_side_len) / resize_h
|
||||
|
||||
resize_h = int(resize_h * ratio)
|
||||
resize_w = int(resize_w * ratio)
|
||||
|
||||
max_stride = 128
|
||||
resize_h = (resize_h + max_stride - 1) // max_stride * max_stride
|
||||
resize_w = (resize_w + max_stride - 1) // max_stride * max_stride
|
||||
im = cv2.resize(im, (int(resize_w), int(resize_h)))
|
||||
ratio_h = resize_h / float(h)
|
||||
ratio_w = resize_w / float(w)
|
||||
return im, (ratio_h, ratio_w)
|
||||
|
||||
|
||||
def point_pair2poly(point_pair_list):
|
||||
"""
|
||||
Transfer vertical point_pairs into poly point in clockwise.
|
||||
"""
|
||||
pair_length_list = []
|
||||
for point_pair in point_pair_list:
|
||||
pair_length = np.linalg.norm(point_pair[0] - point_pair[1])
|
||||
pair_length_list.append(pair_length)
|
||||
pair_length_list = np.array(pair_length_list)
|
||||
pair_info = (
|
||||
pair_length_list.max(),
|
||||
pair_length_list.min(),
|
||||
pair_length_list.mean(),
|
||||
)
|
||||
|
||||
point_num = len(point_pair_list) * 2
|
||||
point_list = [0] * point_num
|
||||
for idx, point_pair in enumerate(point_pair_list):
|
||||
point_list[idx] = point_pair[0]
|
||||
point_list[point_num - 1 - idx] = point_pair[1]
|
||||
return np.array(point_list).reshape(-1, 2), pair_info
|
||||
|
||||
|
||||
def shrink_quad_along_width(quad, begin_width_ratio=0.0, end_width_ratio=1.0):
|
||||
"""
|
||||
Generate shrink_quad_along_width.
|
||||
"""
|
||||
ratio_pair = np.array([[begin_width_ratio], [end_width_ratio]], dtype=np.float32)
|
||||
p0_1 = quad[0] + (quad[1] - quad[0]) * ratio_pair
|
||||
p3_2 = quad[3] + (quad[2] - quad[3]) * ratio_pair
|
||||
return np.array([p0_1[0], p0_1[1], p3_2[1], p3_2[0]])
|
||||
|
||||
|
||||
def expand_poly_along_width(poly, shrink_ratio_of_width=0.3):
|
||||
"""
|
||||
expand poly along width.
|
||||
"""
|
||||
point_num = poly.shape[0]
|
||||
left_quad = np.array([poly[0], poly[1], poly[-2], poly[-1]], dtype=np.float32)
|
||||
left_ratio = (
|
||||
-shrink_ratio_of_width
|
||||
* np.linalg.norm(left_quad[0] - left_quad[3])
|
||||
/ (np.linalg.norm(left_quad[0] - left_quad[1]) + 1e-6)
|
||||
)
|
||||
left_quad_expand = shrink_quad_along_width(left_quad, left_ratio, 1.0)
|
||||
right_quad = np.array(
|
||||
[
|
||||
poly[point_num // 2 - 2],
|
||||
poly[point_num // 2 - 1],
|
||||
poly[point_num // 2],
|
||||
poly[point_num // 2 + 1],
|
||||
],
|
||||
dtype=np.float32,
|
||||
)
|
||||
right_ratio = 1.0 + shrink_ratio_of_width * np.linalg.norm(
|
||||
right_quad[0] - right_quad[3]
|
||||
) / (np.linalg.norm(right_quad[0] - right_quad[1]) + 1e-6)
|
||||
right_quad_expand = shrink_quad_along_width(right_quad, 0.0, right_ratio)
|
||||
poly[0] = left_quad_expand[0]
|
||||
poly[-1] = left_quad_expand[-1]
|
||||
poly[point_num // 2 - 1] = right_quad_expand[1]
|
||||
poly[point_num // 2] = right_quad_expand[2]
|
||||
return poly
|
||||
|
||||
|
||||
def norm2(x, axis=None):
|
||||
if axis:
|
||||
return np.sqrt(np.sum(x**2, axis=axis))
|
||||
return np.sqrt(np.sum(x**2))
|
||||
|
||||
|
||||
def cos(p1, p2):
|
||||
return (p1 * p2).sum() / (norm2(p1) * norm2(p2))
|
||||
95
ppocr/utils/en_dict.txt
Normal file
95
ppocr/utils/en_dict.txt
Normal file
@@ -0,0 +1,95 @@
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
:
|
||||
;
|
||||
<
|
||||
=
|
||||
>
|
||||
?
|
||||
@
|
||||
A
|
||||
B
|
||||
C
|
||||
D
|
||||
E
|
||||
F
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
M
|
||||
N
|
||||
O
|
||||
P
|
||||
Q
|
||||
R
|
||||
S
|
||||
T
|
||||
U
|
||||
V
|
||||
W
|
||||
X
|
||||
Y
|
||||
Z
|
||||
[
|
||||
\
|
||||
]
|
||||
^
|
||||
_
|
||||
`
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
{
|
||||
|
|
||||
}
|
||||
~
|
||||
!
|
||||
"
|
||||
#
|
||||
$
|
||||
%
|
||||
&
|
||||
'
|
||||
(
|
||||
)
|
||||
*
|
||||
+
|
||||
,
|
||||
-
|
||||
.
|
||||
/
|
||||
|
||||
551
ppocr/utils/export_model.py
Normal file
551
ppocr/utils/export_model.py
Normal file
@@ -0,0 +1,551 @@
|
||||
# Copyright (c) 2024 PaddlePaddle Authors. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import os
|
||||
import yaml
|
||||
import json
|
||||
import copy
|
||||
import shutil
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
from paddle.jit import to_static
|
||||
|
||||
from collections import OrderedDict
|
||||
from packaging import version
|
||||
from argparse import ArgumentParser, RawDescriptionHelpFormatter
|
||||
from ppocr.modeling.architectures import build_model
|
||||
from ppocr.postprocess import build_post_process
|
||||
from ppocr.utils.save_load import load_model
|
||||
from ppocr.utils.logging import get_logger
|
||||
|
||||
|
||||
def represent_dictionary_order(self, dict_data):
|
||||
return self.represent_mapping("tag:yaml.org,2002:map", dict_data.items())
|
||||
|
||||
|
||||
def setup_orderdict():
|
||||
yaml.add_representer(OrderedDict, represent_dictionary_order)
|
||||
|
||||
|
||||
def dump_infer_config(config, path, logger):
|
||||
setup_orderdict()
|
||||
infer_cfg = OrderedDict()
|
||||
if not os.path.exists(os.path.dirname(path)):
|
||||
os.makedirs(os.path.dirname(path))
|
||||
model_name = None
|
||||
if config["Global"].get("model_name", None):
|
||||
model_name = config["Global"]["model_name"]
|
||||
infer_cfg["Global"] = {"model_name": model_name}
|
||||
if config["Global"].get("uniform_output_enabled", True):
|
||||
arch_config = config["Architecture"]
|
||||
if arch_config["algorithm"] in ["SVTR_LCNet", "SVTR_HGNet"]:
|
||||
common_dynamic_shapes = {
|
||||
"x": [[1, 3, 48, 160], [1, 3, 48, 320], [8, 3, 48, 3200]]
|
||||
}
|
||||
elif arch_config["model_type"] == "det":
|
||||
common_dynamic_shapes = {
|
||||
"x": [[1, 3, 32, 32], [1, 3, 736, 736], [1, 3, 4000, 4000]]
|
||||
}
|
||||
elif arch_config["algorithm"] == "SLANet":
|
||||
if model_name == "SLANet_plus":
|
||||
common_dynamic_shapes = {
|
||||
"x": [[1, 3, 32, 32], [1, 3, 64, 448], [1, 3, 488, 488]]
|
||||
}
|
||||
else:
|
||||
common_dynamic_shapes = {
|
||||
"x": [[1, 3, 32, 32], [1, 3, 64, 448], [8, 3, 488, 488]]
|
||||
}
|
||||
elif arch_config["algorithm"] == "SLANeXt":
|
||||
common_dynamic_shapes = {
|
||||
"x": [[1, 3, 512, 512], [1, 3, 512, 512], [1, 3, 512, 512]]
|
||||
}
|
||||
elif arch_config["algorithm"] == "LaTeXOCR":
|
||||
common_dynamic_shapes = {
|
||||
"x": [[1, 1, 32, 32], [1, 1, 64, 448], [1, 1, 192, 672]]
|
||||
}
|
||||
elif arch_config["algorithm"] == "UniMERNet":
|
||||
common_dynamic_shapes = {
|
||||
"x": [[1, 1, 192, 672], [1, 1, 192, 672], [8, 1, 192, 672]]
|
||||
}
|
||||
elif arch_config["algorithm"] in ["PP-FormulaNet-L", "PP-FormulaNet_plus-L"]:
|
||||
common_dynamic_shapes = {
|
||||
"x": [[1, 1, 768, 768], [1, 1, 768, 768], [8, 1, 768, 768]]
|
||||
}
|
||||
elif arch_config["algorithm"] in [
|
||||
"PP-FormulaNet-S",
|
||||
"PP-FormulaNet_plus-S",
|
||||
"PP-FormulaNet_plus-M",
|
||||
]:
|
||||
common_dynamic_shapes = {
|
||||
"x": [[1, 1, 384, 384], [1, 1, 384, 384], [8, 1, 384, 384]]
|
||||
}
|
||||
else:
|
||||
common_dynamic_shapes = None
|
||||
|
||||
backend_keys = ["paddle_infer", "tensorrt"]
|
||||
hpi_config = {
|
||||
"backend_configs": {
|
||||
key: {
|
||||
(
|
||||
"dynamic_shapes" if key == "tensorrt" else "trt_dynamic_shapes"
|
||||
): common_dynamic_shapes
|
||||
}
|
||||
for key in backend_keys
|
||||
}
|
||||
}
|
||||
if common_dynamic_shapes:
|
||||
infer_cfg["Hpi"] = hpi_config
|
||||
|
||||
infer_cfg["PreProcess"] = {"transform_ops": config["Eval"]["dataset"]["transforms"]}
|
||||
postprocess = OrderedDict()
|
||||
for k, v in config["PostProcess"].items():
|
||||
if config["Architecture"].get("algorithm") in [
|
||||
"LaTeXOCR",
|
||||
"UniMERNet",
|
||||
"PP-FormulaNet-L",
|
||||
"PP-FormulaNet-S",
|
||||
"PP-FormulaNet_plus-L",
|
||||
"PP-FormulaNet_plus-M",
|
||||
"PP-FormulaNet_plus-S",
|
||||
]:
|
||||
if k != "rec_char_dict_path":
|
||||
postprocess[k] = v
|
||||
else:
|
||||
postprocess[k] = v
|
||||
|
||||
if config["Architecture"].get("algorithm") in ["LaTeXOCR"]:
|
||||
tokenizer_file = config["Global"].get("rec_char_dict_path")
|
||||
if tokenizer_file is not None:
|
||||
with open(tokenizer_file, encoding="utf-8") as tokenizer_config_handle:
|
||||
character_dict = json.load(tokenizer_config_handle)
|
||||
postprocess["character_dict"] = character_dict
|
||||
elif config["Architecture"].get("algorithm") in [
|
||||
"UniMERNet",
|
||||
"PP-FormulaNet-L",
|
||||
"PP-FormulaNet-S",
|
||||
"PP-FormulaNet_plus-L",
|
||||
"PP-FormulaNet_plus-M",
|
||||
"PP-FormulaNet_plus-S",
|
||||
]:
|
||||
tokenizer_file = config["Global"].get("rec_char_dict_path")
|
||||
fast_tokenizer_file = os.path.join(tokenizer_file, "tokenizer.json")
|
||||
tokenizer_config_file = os.path.join(tokenizer_file, "tokenizer_config.json")
|
||||
postprocess["character_dict"] = {}
|
||||
if fast_tokenizer_file is not None:
|
||||
with open(fast_tokenizer_file, encoding="utf-8") as tokenizer_config_handle:
|
||||
character_dict = json.load(tokenizer_config_handle)
|
||||
postprocess["character_dict"]["fast_tokenizer_file"] = character_dict
|
||||
if tokenizer_config_file is not None:
|
||||
with open(
|
||||
tokenizer_config_file, encoding="utf-8"
|
||||
) as tokenizer_config_handle:
|
||||
character_dict = json.load(tokenizer_config_handle)
|
||||
postprocess["character_dict"]["tokenizer_config_file"] = character_dict
|
||||
else:
|
||||
if config["Global"].get("character_dict_path") is not None:
|
||||
with open(config["Global"]["character_dict_path"], encoding="utf-8") as f:
|
||||
lines = f.readlines()
|
||||
character_dict = [line.strip("\n") for line in lines]
|
||||
postprocess["character_dict"] = character_dict
|
||||
|
||||
infer_cfg["PostProcess"] = postprocess
|
||||
|
||||
with open(path, "w", encoding="utf-8") as f:
|
||||
yaml.dump(infer_cfg, f, default_flow_style=False, allow_unicode=True)
|
||||
logger.info("Export inference config file to {}".format(os.path.join(path)))
|
||||
|
||||
|
||||
def dynamic_to_static(model, arch_config, logger, input_shape=None):
|
||||
if arch_config["algorithm"] == "SRN":
|
||||
max_text_length = arch_config["Head"]["max_text_length"]
|
||||
other_shape = [
|
||||
paddle.static.InputSpec(shape=[None, 1, 64, 256], dtype="float32"),
|
||||
[
|
||||
paddle.static.InputSpec(shape=[None, 256, 1], dtype="int64"),
|
||||
paddle.static.InputSpec(
|
||||
shape=[None, max_text_length, 1], dtype="int64"
|
||||
),
|
||||
paddle.static.InputSpec(
|
||||
shape=[None, 8, max_text_length, max_text_length], dtype="int64"
|
||||
),
|
||||
paddle.static.InputSpec(
|
||||
shape=[None, 8, max_text_length, max_text_length], dtype="int64"
|
||||
),
|
||||
],
|
||||
]
|
||||
model = to_static(model, input_spec=other_shape)
|
||||
elif arch_config["algorithm"] == "SAR":
|
||||
other_shape = [
|
||||
paddle.static.InputSpec(shape=[None, 3, 48, 160], dtype="float32"),
|
||||
[paddle.static.InputSpec(shape=[None], dtype="float32")],
|
||||
]
|
||||
model = to_static(model, input_spec=other_shape)
|
||||
elif arch_config["algorithm"] in ["SVTR_LCNet", "SVTR_HGNet"]:
|
||||
other_shape = [
|
||||
paddle.static.InputSpec(shape=[None, 3, 48, -1], dtype="float32"),
|
||||
]
|
||||
model = to_static(model, input_spec=other_shape)
|
||||
elif arch_config["algorithm"] in ["SVTR", "CPPD"]:
|
||||
other_shape = [
|
||||
paddle.static.InputSpec(shape=[None] + input_shape, dtype="float32"),
|
||||
]
|
||||
model = to_static(model, input_spec=other_shape)
|
||||
elif arch_config["algorithm"] == "PREN":
|
||||
other_shape = [
|
||||
paddle.static.InputSpec(shape=[None, 3, 64, 256], dtype="float32"),
|
||||
]
|
||||
model = to_static(model, input_spec=other_shape)
|
||||
elif arch_config["model_type"] == "sr":
|
||||
other_shape = [
|
||||
paddle.static.InputSpec(shape=[None, 3, 16, 64], dtype="float32")
|
||||
]
|
||||
model = to_static(model, input_spec=other_shape)
|
||||
elif arch_config["algorithm"] == "ViTSTR":
|
||||
other_shape = [
|
||||
paddle.static.InputSpec(shape=[None, 1, 224, 224], dtype="float32"),
|
||||
]
|
||||
model = to_static(model, input_spec=other_shape)
|
||||
elif arch_config["algorithm"] == "ABINet":
|
||||
if not input_shape:
|
||||
input_shape = [3, 32, 128]
|
||||
other_shape = [
|
||||
paddle.static.InputSpec(shape=[None] + input_shape, dtype="float32"),
|
||||
]
|
||||
model = to_static(model, input_spec=other_shape)
|
||||
elif arch_config["algorithm"] in ["NRTR", "SPIN", "RFL"]:
|
||||
other_shape = [
|
||||
paddle.static.InputSpec(shape=[None, 1, 32, 100], dtype="float32"),
|
||||
]
|
||||
model = to_static(model, input_spec=other_shape)
|
||||
elif arch_config["algorithm"] in ["SATRN"]:
|
||||
other_shape = [
|
||||
paddle.static.InputSpec(shape=[None, 3, 32, 100], dtype="float32"),
|
||||
]
|
||||
model = to_static(model, input_spec=other_shape)
|
||||
elif arch_config["algorithm"] == "VisionLAN":
|
||||
other_shape = [
|
||||
paddle.static.InputSpec(shape=[None, 3, 64, 256], dtype="float32"),
|
||||
]
|
||||
model = to_static(model, input_spec=other_shape)
|
||||
elif arch_config["algorithm"] == "RobustScanner":
|
||||
max_text_length = arch_config["Head"]["max_text_length"]
|
||||
other_shape = [
|
||||
paddle.static.InputSpec(shape=[None, 3, 48, 160], dtype="float32"),
|
||||
[
|
||||
paddle.static.InputSpec(
|
||||
shape=[
|
||||
None,
|
||||
],
|
||||
dtype="float32",
|
||||
),
|
||||
paddle.static.InputSpec(shape=[None, max_text_length], dtype="int64"),
|
||||
],
|
||||
]
|
||||
model = to_static(model, input_spec=other_shape)
|
||||
elif arch_config["algorithm"] == "CAN":
|
||||
other_shape = [
|
||||
[
|
||||
paddle.static.InputSpec(shape=[None, 1, None, None], dtype="float32"),
|
||||
paddle.static.InputSpec(shape=[None, 1, None, None], dtype="float32"),
|
||||
paddle.static.InputSpec(
|
||||
shape=[None, arch_config["Head"]["max_text_length"]], dtype="int64"
|
||||
),
|
||||
]
|
||||
]
|
||||
model = to_static(model, input_spec=other_shape)
|
||||
elif arch_config["algorithm"] == "LaTeXOCR":
|
||||
other_shape = [
|
||||
paddle.static.InputSpec(shape=[None, 1, None, None], dtype="float32"),
|
||||
]
|
||||
model = to_static(model, input_spec=other_shape)
|
||||
elif arch_config["algorithm"] == "UniMERNet":
|
||||
model = paddle.jit.to_static(
|
||||
model,
|
||||
input_spec=[
|
||||
paddle.static.InputSpec(shape=[-1, 1, 192, 672], dtype="float32")
|
||||
],
|
||||
full_graph=True,
|
||||
)
|
||||
elif arch_config["algorithm"] == "SLANeXt":
|
||||
model = paddle.jit.to_static(
|
||||
model,
|
||||
input_spec=[
|
||||
paddle.static.InputSpec(shape=[-1, 3, 512, 512], dtype="float32")
|
||||
],
|
||||
full_graph=True,
|
||||
)
|
||||
elif arch_config["algorithm"] in ["PP-FormulaNet-L", "PP-FormulaNet_plus-L"]:
|
||||
model = paddle.jit.to_static(
|
||||
model,
|
||||
input_spec=[
|
||||
paddle.static.InputSpec(shape=[-1, 1, 768, 768], dtype="float32")
|
||||
],
|
||||
full_graph=True,
|
||||
)
|
||||
elif arch_config["algorithm"] in [
|
||||
"PP-FormulaNet-S",
|
||||
"PP-FormulaNet_plus-S",
|
||||
"PP-FormulaNet_plus-M",
|
||||
]:
|
||||
model = paddle.jit.to_static(
|
||||
model,
|
||||
input_spec=[
|
||||
paddle.static.InputSpec(shape=[-1, 1, 384, 384], dtype="float32")
|
||||
],
|
||||
full_graph=True,
|
||||
)
|
||||
|
||||
elif arch_config["algorithm"] in ["LayoutLM", "LayoutLMv2", "LayoutXLM"]:
|
||||
input_spec = [
|
||||
paddle.static.InputSpec(shape=[None, 512], dtype="int64"), # input_ids
|
||||
paddle.static.InputSpec(shape=[None, 512, 4], dtype="int64"), # bbox
|
||||
paddle.static.InputSpec(shape=[None, 512], dtype="int64"), # attention_mask
|
||||
paddle.static.InputSpec(shape=[None, 512], dtype="int64"), # token_type_ids
|
||||
paddle.static.InputSpec(shape=[None, 3, 224, 224], dtype="int64"), # image
|
||||
]
|
||||
if "Re" in arch_config["Backbone"]["name"]:
|
||||
input_spec.extend(
|
||||
[
|
||||
paddle.static.InputSpec(
|
||||
shape=[None, 512, 3], dtype="int64"
|
||||
), # entities
|
||||
paddle.static.InputSpec(
|
||||
shape=[None, None, 2], dtype="int64"
|
||||
), # relations
|
||||
]
|
||||
)
|
||||
if model.backbone.use_visual_backbone is False:
|
||||
input_spec.pop(4)
|
||||
model = to_static(model, input_spec=[input_spec])
|
||||
else:
|
||||
infer_shape = [3, -1, -1]
|
||||
if arch_config["model_type"] == "rec":
|
||||
infer_shape = [3, 32, -1] # for rec model, H must be 32
|
||||
if (
|
||||
"Transform" in arch_config
|
||||
and arch_config["Transform"] is not None
|
||||
and arch_config["Transform"]["name"] == "TPS"
|
||||
):
|
||||
logger.info(
|
||||
"When there is tps in the network, variable length input is not supported, and the input size needs to be the same as during training"
|
||||
)
|
||||
infer_shape[-1] = 100
|
||||
elif arch_config["model_type"] == "table":
|
||||
infer_shape = [3, 488, 488]
|
||||
if arch_config["algorithm"] == "TableMaster":
|
||||
infer_shape = [3, 480, 480]
|
||||
if arch_config["algorithm"] == "SLANet":
|
||||
infer_shape = [3, -1, -1]
|
||||
model = to_static(
|
||||
model,
|
||||
input_spec=[
|
||||
paddle.static.InputSpec(shape=[None] + infer_shape, dtype="float32")
|
||||
],
|
||||
)
|
||||
|
||||
if (
|
||||
arch_config["model_type"] != "sr"
|
||||
and arch_config["Backbone"]["name"] == "PPLCNetV3"
|
||||
):
|
||||
# for rep lcnetv3
|
||||
for layer in model.sublayers():
|
||||
if hasattr(layer, "rep") and not getattr(layer, "is_repped"):
|
||||
layer.rep()
|
||||
return model
|
||||
|
||||
|
||||
def export_single_model(
|
||||
model,
|
||||
arch_config,
|
||||
save_path,
|
||||
logger,
|
||||
yaml_path,
|
||||
config,
|
||||
input_shape=None,
|
||||
quanter=None,
|
||||
):
|
||||
|
||||
model = dynamic_to_static(model, arch_config, logger, input_shape)
|
||||
|
||||
if quanter is None:
|
||||
try:
|
||||
import encryption # Attempt to import the encryption module for AIStudio's encryption model
|
||||
except (
|
||||
ModuleNotFoundError
|
||||
): # Encryption is not needed if the module cannot be imported
|
||||
print("Skipping import of the encryption module")
|
||||
paddle_version = version.parse(paddle.__version__)
|
||||
if config["Global"].get("export_with_pir", True):
|
||||
assert (
|
||||
paddle_version >= version.parse("3.0.0b2")
|
||||
or paddle_version == version.parse("0.0.0")
|
||||
) and os.environ.get("FLAGS_enable_pir_api", None) not in ["0", "False"]
|
||||
paddle.jit.save(model, save_path)
|
||||
else:
|
||||
if paddle_version >= version.parse(
|
||||
"3.0.0b2"
|
||||
) or paddle_version == version.parse("0.0.0"):
|
||||
model.forward.rollback()
|
||||
with paddle.pir_utils.OldIrGuard():
|
||||
model = dynamic_to_static(model, arch_config, logger, input_shape)
|
||||
paddle.jit.save(model, save_path)
|
||||
else:
|
||||
paddle.jit.save(model, save_path)
|
||||
else:
|
||||
quanter.save_quantized_model(model, save_path)
|
||||
logger.info("inference model is saved to {}".format(save_path))
|
||||
return
|
||||
|
||||
|
||||
def convert_bn(model):
|
||||
for n, m in model.named_children():
|
||||
if isinstance(m, nn.SyncBatchNorm):
|
||||
bn = nn.BatchNorm2D(
|
||||
m._num_features, m._momentum, m._epsilon, m._weight_attr, m._bias_attr
|
||||
)
|
||||
bn.set_dict(m.state_dict())
|
||||
setattr(model, n, bn)
|
||||
else:
|
||||
convert_bn(m)
|
||||
|
||||
|
||||
def export(config, base_model=None, save_path=None):
|
||||
if paddle.distributed.get_rank() != 0:
|
||||
return
|
||||
logger = get_logger()
|
||||
# build post process
|
||||
post_process_class = build_post_process(config["PostProcess"], config["Global"])
|
||||
|
||||
# build model
|
||||
# for rec algorithm
|
||||
if hasattr(post_process_class, "character"):
|
||||
char_num = len(getattr(post_process_class, "character"))
|
||||
if config["Architecture"]["algorithm"] in [
|
||||
"Distillation",
|
||||
]: # distillation model
|
||||
for key in config["Architecture"]["Models"]:
|
||||
if (
|
||||
config["Architecture"]["Models"][key]["Head"]["name"] == "MultiHead"
|
||||
): # multi head
|
||||
out_channels_list = {}
|
||||
if config["PostProcess"]["name"] == "DistillationSARLabelDecode":
|
||||
char_num = char_num - 2
|
||||
if config["PostProcess"]["name"] == "DistillationNRTRLabelDecode":
|
||||
char_num = char_num - 3
|
||||
out_channels_list["CTCLabelDecode"] = char_num
|
||||
out_channels_list["SARLabelDecode"] = char_num + 2
|
||||
out_channels_list["NRTRLabelDecode"] = char_num + 3
|
||||
config["Architecture"]["Models"][key]["Head"][
|
||||
"out_channels_list"
|
||||
] = out_channels_list
|
||||
else:
|
||||
config["Architecture"]["Models"][key]["Head"][
|
||||
"out_channels"
|
||||
] = char_num
|
||||
# just one final tensor needs to exported for inference
|
||||
config["Architecture"]["Models"][key]["return_all_feats"] = False
|
||||
elif config["Architecture"]["Head"]["name"] == "MultiHead": # multi head
|
||||
out_channels_list = {}
|
||||
char_num = len(getattr(post_process_class, "character"))
|
||||
if config["PostProcess"]["name"] == "SARLabelDecode":
|
||||
char_num = char_num - 2
|
||||
if config["PostProcess"]["name"] == "NRTRLabelDecode":
|
||||
char_num = char_num - 3
|
||||
out_channels_list["CTCLabelDecode"] = char_num
|
||||
out_channels_list["SARLabelDecode"] = char_num + 2
|
||||
out_channels_list["NRTRLabelDecode"] = char_num + 3
|
||||
config["Architecture"]["Head"]["out_channels_list"] = out_channels_list
|
||||
else: # base rec model
|
||||
config["Architecture"]["Head"]["out_channels"] = char_num
|
||||
|
||||
# for sr algorithm
|
||||
if config["Architecture"]["model_type"] == "sr":
|
||||
config["Architecture"]["Transform"]["infer_mode"] = True
|
||||
|
||||
# for latexocr algorithm
|
||||
if config["Architecture"].get("algorithm") in ["LaTeXOCR"]:
|
||||
config["Architecture"]["Backbone"]["is_predict"] = True
|
||||
config["Architecture"]["Backbone"]["is_export"] = True
|
||||
config["Architecture"]["Head"]["is_export"] = True
|
||||
if config["Architecture"].get("algorithm") in ["UniMERNet"]:
|
||||
config["Architecture"]["Backbone"]["is_export"] = True
|
||||
config["Architecture"]["Head"]["is_export"] = True
|
||||
if config["Architecture"].get("algorithm") in [
|
||||
"PP-FormulaNet-S",
|
||||
"PP-FormulaNet-L",
|
||||
"PP-FormulaNet_plus-S",
|
||||
"PP-FormulaNet_plus-M",
|
||||
"PP-FormulaNet_plus-L",
|
||||
]:
|
||||
config["Architecture"]["Head"]["is_export"] = True
|
||||
if base_model is not None:
|
||||
model = base_model
|
||||
if isinstance(model, paddle.DataParallel):
|
||||
model = copy.deepcopy(model._layers)
|
||||
else:
|
||||
model = copy.deepcopy(model)
|
||||
else:
|
||||
model = build_model(config["Architecture"])
|
||||
load_model(config, model, model_type=config["Architecture"]["model_type"])
|
||||
convert_bn(model)
|
||||
model.eval()
|
||||
|
||||
if not save_path:
|
||||
save_path = config["Global"]["save_inference_dir"]
|
||||
yaml_path = os.path.join(save_path, "inference.yml")
|
||||
|
||||
arch_config = config["Architecture"]
|
||||
|
||||
if (
|
||||
arch_config["algorithm"] in ["SVTR", "CPPD"]
|
||||
and arch_config["Head"]["name"] != "MultiHead"
|
||||
):
|
||||
input_shape = config["Eval"]["dataset"]["transforms"][-2]["SVTRRecResizeImg"][
|
||||
"image_shape"
|
||||
]
|
||||
elif arch_config["algorithm"].lower() == "ABINet".lower():
|
||||
rec_rs = [
|
||||
c
|
||||
for c in config["Eval"]["dataset"]["transforms"]
|
||||
if "ABINetRecResizeImg" in c
|
||||
]
|
||||
input_shape = rec_rs[0]["ABINetRecResizeImg"]["image_shape"] if rec_rs else None
|
||||
else:
|
||||
input_shape = None
|
||||
dump_infer_config(config, yaml_path, logger)
|
||||
if arch_config["algorithm"] in [
|
||||
"Distillation",
|
||||
]: # distillation model
|
||||
archs = list(arch_config["Models"].values())
|
||||
for idx, name in enumerate(model.model_name_list):
|
||||
sub_model_save_path = os.path.join(save_path, name, "inference")
|
||||
export_single_model(
|
||||
model.model_list[idx],
|
||||
archs[idx],
|
||||
sub_model_save_path,
|
||||
logger,
|
||||
yaml_path,
|
||||
config,
|
||||
)
|
||||
else:
|
||||
save_path = os.path.join(save_path, "inference")
|
||||
export_single_model(
|
||||
model,
|
||||
arch_config,
|
||||
save_path,
|
||||
logger,
|
||||
yaml_path,
|
||||
config,
|
||||
input_shape=input_shape,
|
||||
)
|
||||
74
ppocr/utils/formula_utils/math_txt2pkl.py
Normal file
74
ppocr/utils/formula_utils/math_txt2pkl.py
Normal file
@@ -0,0 +1,74 @@
|
||||
# copyright (c) 2024 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import pickle
|
||||
from tqdm import tqdm
|
||||
import os
|
||||
import math
|
||||
from paddle.utils import try_import
|
||||
from collections import defaultdict
|
||||
import glob
|
||||
from os.path import join
|
||||
import argparse
|
||||
|
||||
|
||||
def txt2pickle(images, equations, save_dir):
|
||||
imagesize = try_import("imagesize")
|
||||
save_p = os.path.join(save_dir, "latexocr_{}.pkl".format(images.split("/")[-1]))
|
||||
min_dimensions = (32, 32)
|
||||
max_dimensions = (672, 192)
|
||||
max_length = 512
|
||||
data = defaultdict(lambda: [])
|
||||
if images is not None and equations is not None:
|
||||
images_list = [
|
||||
path.replace("\\", "/") for path in glob.glob(join(images, "*.png"))
|
||||
]
|
||||
indices = [int(os.path.basename(img).split(".")[0]) for img in images_list]
|
||||
eqs = open(equations, "r").read().split("\n")
|
||||
for i, im in tqdm(enumerate(images_list), total=len(images_list)):
|
||||
width, height = imagesize.get(im)
|
||||
if (
|
||||
min_dimensions[0] <= width <= max_dimensions[0]
|
||||
and min_dimensions[1] <= height <= max_dimensions[1]
|
||||
):
|
||||
divide_h = math.ceil(height / 16) * 16
|
||||
divide_w = math.ceil(width / 16) * 16
|
||||
im = os.path.basename(im)
|
||||
data[(divide_w, divide_h)].append((eqs[indices[i]], im))
|
||||
data = dict(data)
|
||||
with open(save_p, "wb") as file:
|
||||
pickle.dump(data, file)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser()
|
||||
|
||||
parser.add_argument(
|
||||
"--image_dir",
|
||||
type=str,
|
||||
default=".",
|
||||
help="Input_label or input path to be converted",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--mathtxt_path",
|
||||
type=str,
|
||||
default=".",
|
||||
help="Input_label or input path to be converted",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--output_dir", type=str, default="out_label.txt", help="Output file name"
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
txt2pickle(args.image_dir, args.mathtxt_path, args.output_dir)
|
||||
107
ppocr/utils/formula_utils/unimernet_data_convert.py
Normal file
107
ppocr/utils/formula_utils/unimernet_data_convert.py
Normal file
@@ -0,0 +1,107 @@
|
||||
# copyright (c) 2024 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import os
|
||||
import cv2
|
||||
import glob
|
||||
import argparse
|
||||
from os.path import join
|
||||
from tqdm import tqdm
|
||||
|
||||
|
||||
def latexocr2paddleocr_train(image_path, math_unimernet_file, math_hwe_file, save_path):
|
||||
convert_f = open(save_path, "w")
|
||||
sub_dir = "UniMER-1M/images"
|
||||
img_sub_dir = os.path.join(image_path, sub_dir)
|
||||
with open(math_unimernet_file, "r") as f:
|
||||
lines = f.readlines()
|
||||
formula_num = len(lines)
|
||||
for i, line in tqdm(enumerate(lines), total=formula_num):
|
||||
image_name = "{0:07d}.png".format(i)
|
||||
math_gt = line.strip()
|
||||
image_p = os.path.join(img_sub_dir, image_name)
|
||||
img_name_subdir = os.path.join(sub_dir, image_name)
|
||||
if os.path.exists(image_p):
|
||||
convert_f.writelines("{}\t{}\n".format(img_name_subdir, math_gt))
|
||||
|
||||
sub_dir = "HME100K/train_images"
|
||||
img_sub_dir = os.path.join(image_path, sub_dir)
|
||||
with open(math_hwe_file, "r") as f:
|
||||
lines = f.readlines()
|
||||
formula_num = len(lines)
|
||||
for i, line in tqdm(enumerate(lines), total=formula_num):
|
||||
img_name, math_gt = line.strip().split("\t")
|
||||
image_path = os.path.join(img_sub_dir, img_name)
|
||||
img_name_subdir = os.path.join(sub_dir, img_name)
|
||||
convert_f.writelines("{}\t{}\n".format(img_name_subdir, math_gt))
|
||||
|
||||
convert_f.close()
|
||||
|
||||
|
||||
def unimernet2paddleocr_test(image_path, math_file, save_path):
|
||||
convert_f = open(save_path, "w")
|
||||
with open(math_file, "r") as f:
|
||||
# load maths which
|
||||
lines = f.readlines()
|
||||
formula_num = len(lines)
|
||||
for i, line in tqdm(enumerate(lines), total=formula_num):
|
||||
image_name = "{0:07d}.png".format(i)
|
||||
math_gt = line.strip()
|
||||
image_p = os.path.join(image_path, image_name)
|
||||
if os.path.exists(image_p):
|
||||
convert_f.writelines("{}\t{}\n".format(image_name, math_gt))
|
||||
convert_f.close()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser()
|
||||
|
||||
parser.add_argument(
|
||||
"--image_dir",
|
||||
type=str,
|
||||
default=".",
|
||||
help="Input_label or input path to be converted",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--unimernet_txt_path",
|
||||
type=str,
|
||||
default="",
|
||||
help="Input_label or input path to be converted",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--hme100k_txt_path",
|
||||
type=str,
|
||||
default="",
|
||||
help="Input_label or input path to be converted",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--output_path", type=str, default="out_label.txt", help="Output file name"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--datatype", type=str, default="out_label.txt", help="datatype"
|
||||
)
|
||||
args = parser.parse_args()
|
||||
if args.datatype == "unimernet_train":
|
||||
latexocr2paddleocr_train(
|
||||
args.image_dir,
|
||||
args.unimernet_txt_path,
|
||||
args.hme100k_txt_path,
|
||||
args.output_path,
|
||||
)
|
||||
elif args.datatype == "unimernet_test":
|
||||
unimernet2paddleocr_test(
|
||||
args.image_dir, args.unimernet_txt_path, args.output_path
|
||||
)
|
||||
else:
|
||||
raise NotImplementedError("the datatype is not supported")
|
||||
82
ppocr/utils/gen_label.py
Normal file
82
ppocr/utils/gen_label.py
Normal file
@@ -0,0 +1,82 @@
|
||||
# copyright (c) 2020 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
import os
|
||||
import argparse
|
||||
import json
|
||||
|
||||
|
||||
def gen_rec_label(input_path, out_label):
|
||||
with open(out_label, "w") as out_file:
|
||||
with open(input_path, "r") as f:
|
||||
for line in f.readlines():
|
||||
tmp = line.strip("\n").replace(" ", "").split(",")
|
||||
img_path, label = tmp[0], tmp[1]
|
||||
label = label.replace('"', "")
|
||||
out_file.write(img_path + "\t" + label + "\n")
|
||||
|
||||
|
||||
def gen_det_label(root_path, input_dir, out_label):
|
||||
with open(out_label, "w") as out_file:
|
||||
for label_file in os.listdir(input_dir):
|
||||
img_path = os.path.join(root_path, label_file[3:-4] + ".jpg")
|
||||
label = []
|
||||
with open(
|
||||
os.path.join(input_dir, label_file), "r", encoding="utf-8-sig"
|
||||
) as f:
|
||||
for line in f.readlines():
|
||||
tmp = line.strip("\n\r").replace("\xef\xbb\xbf", "").split(",")
|
||||
points = tmp[:8]
|
||||
s = []
|
||||
for i in range(0, len(points), 2):
|
||||
b = points[i : i + 2]
|
||||
b = [int(t) for t in b]
|
||||
s.append(b)
|
||||
result = {"transcription": tmp[8], "points": s}
|
||||
label.append(result)
|
||||
|
||||
out_file.write(
|
||||
img_path + "\t" + json.dumps(label, ensure_ascii=False) + "\n"
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument(
|
||||
"--mode",
|
||||
type=str,
|
||||
default="rec",
|
||||
help="Generate rec_label or det_label, can be set rec or det",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--root_path",
|
||||
type=str,
|
||||
default=".",
|
||||
help="The root directory of images.Only takes effect when mode=det ",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--input_path",
|
||||
type=str,
|
||||
default=".",
|
||||
help="Input_label or input path to be converted",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--output_label", type=str, default="out_label.txt", help="Output file name"
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
if args.mode == "rec":
|
||||
print("Generate rec label")
|
||||
gen_rec_label(args.input_path, args.output_label)
|
||||
elif args.mode == "det":
|
||||
gen_det_label(args.root_path, args.input_path, args.output_label)
|
||||
36
ppocr/utils/ic15_dict.txt
Normal file
36
ppocr/utils/ic15_dict.txt
Normal file
@@ -0,0 +1,36 @@
|
||||
0
|
||||
1
|
||||
2
|
||||
3
|
||||
4
|
||||
5
|
||||
6
|
||||
7
|
||||
8
|
||||
9
|
||||
a
|
||||
b
|
||||
c
|
||||
d
|
||||
e
|
||||
f
|
||||
g
|
||||
h
|
||||
i
|
||||
j
|
||||
k
|
||||
l
|
||||
m
|
||||
n
|
||||
o
|
||||
p
|
||||
q
|
||||
r
|
||||
s
|
||||
t
|
||||
u
|
||||
v
|
||||
w
|
||||
x
|
||||
y
|
||||
z
|
||||
54
ppocr/utils/iou.py
Normal file
54
ppocr/utils/iou.py
Normal file
@@ -0,0 +1,54 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/whai362/PSENet/blob/python3/models/loss/iou.py
|
||||
"""
|
||||
|
||||
import paddle
|
||||
|
||||
EPS = 1e-6
|
||||
|
||||
|
||||
def iou_single(a, b, mask, n_class):
|
||||
valid = mask == 1
|
||||
a = a.masked_select(valid)
|
||||
b = b.masked_select(valid)
|
||||
miou = []
|
||||
for i in range(n_class):
|
||||
if a.shape == [0] and a.shape == b.shape:
|
||||
inter = paddle.to_tensor(0.0)
|
||||
union = paddle.to_tensor(0.0)
|
||||
else:
|
||||
inter = ((a == i).logical_and(b == i)).astype("float32")
|
||||
union = ((a == i).logical_or(b == i)).astype("float32")
|
||||
miou.append(paddle.sum(inter) / (paddle.sum(union) + EPS))
|
||||
miou = sum(miou) / len(miou)
|
||||
return miou
|
||||
|
||||
|
||||
def iou(a, b, mask, n_class=2, reduce=True):
|
||||
batch_size = a.shape[0]
|
||||
|
||||
a = a.reshape([batch_size, -1])
|
||||
b = b.reshape([batch_size, -1])
|
||||
mask = mask.reshape([batch_size, -1])
|
||||
|
||||
iou = paddle.zeros((batch_size,), dtype="float32")
|
||||
for i in range(batch_size):
|
||||
iou[i] = iou_single(a[i], b[i], mask[i], n_class)
|
||||
|
||||
if reduce:
|
||||
iou = paddle.mean(iou)
|
||||
return iou
|
||||
2
ppocr/utils/loggers/__init__.py
Normal file
2
ppocr/utils/loggers/__init__.py
Normal file
@@ -0,0 +1,2 @@
|
||||
from .wandb_logger import WandbLogger
|
||||
from .loggers import Loggers
|
||||
16
ppocr/utils/loggers/base_logger.py
Normal file
16
ppocr/utils/loggers/base_logger.py
Normal file
@@ -0,0 +1,16 @@
|
||||
import os
|
||||
from abc import ABC, abstractmethod
|
||||
|
||||
|
||||
class BaseLogger(ABC):
|
||||
def __init__(self, save_dir):
|
||||
self.save_dir = save_dir
|
||||
os.makedirs(self.save_dir, exist_ok=True)
|
||||
|
||||
@abstractmethod
|
||||
def log_metrics(self, metrics, prefix=None):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def close(self):
|
||||
pass
|
||||
19
ppocr/utils/loggers/loggers.py
Normal file
19
ppocr/utils/loggers/loggers.py
Normal file
@@ -0,0 +1,19 @@
|
||||
from .wandb_logger import WandbLogger
|
||||
|
||||
|
||||
class Loggers(object):
|
||||
def __init__(self, loggers):
|
||||
super().__init__()
|
||||
self.loggers = loggers
|
||||
|
||||
def log_metrics(self, metrics, prefix=None, step=None):
|
||||
for logger in self.loggers:
|
||||
logger.log_metrics(metrics, prefix=prefix, step=step)
|
||||
|
||||
def log_model(self, is_best, prefix, metadata=None):
|
||||
for logger in self.loggers:
|
||||
logger.log_model(is_best=is_best, prefix=prefix, metadata=metadata)
|
||||
|
||||
def close(self):
|
||||
for logger in self.loggers:
|
||||
logger.close()
|
||||
84
ppocr/utils/loggers/wandb_logger.py
Normal file
84
ppocr/utils/loggers/wandb_logger.py
Normal file
@@ -0,0 +1,84 @@
|
||||
import os
|
||||
from .base_logger import BaseLogger
|
||||
from ppocr.utils.logging import get_logger
|
||||
|
||||
|
||||
class WandbLogger(BaseLogger):
|
||||
def __init__(
|
||||
self,
|
||||
project=None,
|
||||
name=None,
|
||||
id=None,
|
||||
entity=None,
|
||||
save_dir=None,
|
||||
config=None,
|
||||
**kwargs,
|
||||
):
|
||||
try:
|
||||
import wandb
|
||||
|
||||
self.wandb = wandb
|
||||
except ModuleNotFoundError:
|
||||
raise ModuleNotFoundError("Please install wandb using `pip install wandb`")
|
||||
|
||||
self.project = project
|
||||
self.name = name
|
||||
self.id = id
|
||||
self.save_dir = save_dir
|
||||
self.config = config
|
||||
self.kwargs = kwargs
|
||||
self.entity = entity
|
||||
self._run = None
|
||||
self._wandb_init = dict(
|
||||
project=self.project,
|
||||
name=self.name,
|
||||
id=self.id,
|
||||
entity=self.entity,
|
||||
dir=self.save_dir,
|
||||
resume="allow",
|
||||
)
|
||||
self._wandb_init.update(**kwargs)
|
||||
self.logger = get_logger()
|
||||
|
||||
_ = self.run
|
||||
|
||||
if self.config:
|
||||
self.run.config.update(self.config)
|
||||
|
||||
@property
|
||||
def run(self):
|
||||
if self._run is None:
|
||||
if self.wandb.run is not None:
|
||||
self.logger.info(
|
||||
"There is a wandb run already in progress "
|
||||
"and newly created instances of `WandbLogger` will reuse"
|
||||
" this run. If this is not desired, call `wandb.finish()`"
|
||||
"before instantiating `WandbLogger`."
|
||||
)
|
||||
self._run = self.wandb.run
|
||||
else:
|
||||
self._run = self.wandb.init(**self._wandb_init)
|
||||
return self._run
|
||||
|
||||
def log_metrics(self, metrics, prefix=None, step=None):
|
||||
if not prefix:
|
||||
prefix = ""
|
||||
updated_metrics = {prefix.lower() + "/" + k: v for k, v in metrics.items()}
|
||||
|
||||
self.run.log(updated_metrics, step=step)
|
||||
|
||||
def log_model(self, is_best, prefix, metadata=None):
|
||||
model_path = os.path.join(self.save_dir, prefix + ".pdparams")
|
||||
artifact = self.wandb.Artifact(
|
||||
"model-{}".format(self.run.id), type="model", metadata=metadata
|
||||
)
|
||||
artifact.add_file(model_path, name="model_ckpt.pdparams")
|
||||
|
||||
aliases = [prefix]
|
||||
if is_best:
|
||||
aliases.append("best")
|
||||
|
||||
self.run.log_artifact(artifact, aliases=aliases)
|
||||
|
||||
def close(self):
|
||||
self.run.finish()
|
||||
78
ppocr/utils/logging.py
Normal file
78
ppocr/utils/logging.py
Normal file
@@ -0,0 +1,78 @@
|
||||
# copyright (c) 2020 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/WenmuZhou/PytorchOCR/blob/master/torchocr/utils/logging.py
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import logging
|
||||
import functools
|
||||
import paddle.distributed as dist
|
||||
|
||||
logger_initialized = {}
|
||||
|
||||
|
||||
@functools.lru_cache()
|
||||
def get_logger(name="ppocr", log_file=None, log_level=logging.DEBUG, log_ranks="0"):
|
||||
"""Initialize and get a logger by name.
|
||||
If the logger has not been initialized, this method will initialize the
|
||||
logger by adding one or two handlers, otherwise the initialized logger will
|
||||
be directly returned. During initialization, a StreamHandler will always be
|
||||
added. If `log_file` is specified a FileHandler will also be added.
|
||||
Args:
|
||||
name (str): Logger name.
|
||||
log_file (str | None): The log filename. If specified, a FileHandler
|
||||
will be added to the logger.
|
||||
log_level (int): The logger level. Note that only the process of
|
||||
rank 0 is affected, and other processes will set the level to
|
||||
"Error" thus be silent most of the time.
|
||||
log_ranks (str): The ids of gpu to log which are separated by "," when more than 1, "0" by default.
|
||||
Returns:
|
||||
logging.Logger: The expected logger.
|
||||
"""
|
||||
logger = logging.getLogger(name)
|
||||
if name in logger_initialized:
|
||||
return logger
|
||||
for logger_name in logger_initialized:
|
||||
if name.startswith(logger_name):
|
||||
return logger
|
||||
|
||||
formatter = logging.Formatter(
|
||||
"[%(asctime)s] %(name)s %(levelname)s: %(message)s", datefmt="%Y/%m/%d %H:%M:%S"
|
||||
)
|
||||
|
||||
stream_handler = logging.StreamHandler(stream=sys.stdout)
|
||||
stream_handler.setFormatter(formatter)
|
||||
logger.addHandler(stream_handler)
|
||||
if log_file is not None and dist.get_rank() == 0:
|
||||
log_file_folder = os.path.split(log_file)[0]
|
||||
os.makedirs(log_file_folder, exist_ok=True)
|
||||
file_handler = logging.FileHandler(log_file, "a")
|
||||
file_handler.setFormatter(formatter)
|
||||
logger.addHandler(file_handler)
|
||||
|
||||
if isinstance(log_ranks, str):
|
||||
log_ranks = [int(i) for i in log_ranks.split(",")]
|
||||
elif isinstance(log_ranks, int):
|
||||
log_ranks = [log_ranks]
|
||||
|
||||
if dist.get_rank() in log_ranks:
|
||||
logger.setLevel(log_level)
|
||||
else:
|
||||
logger.setLevel(logging.ERROR)
|
||||
logger_initialized[name] = True
|
||||
logger.propagate = False
|
||||
return logger
|
||||
155
ppocr/utils/network.py
Normal file
155
ppocr/utils/network.py
Normal file
@@ -0,0 +1,155 @@
|
||||
# copyright (c) 2020 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
import shutil
|
||||
import tarfile
|
||||
import requests
|
||||
import os.path as osp
|
||||
import paddle.distributed as dist
|
||||
from tqdm import tqdm
|
||||
|
||||
from ppocr.utils.logging import get_logger
|
||||
|
||||
MODELS_DIR = os.path.join(
|
||||
os.environ.get("PADDLE_OCR_BASE_DIR", os.path.expanduser("~/.paddleocr/")), "models"
|
||||
)
|
||||
DOWNLOAD_RETRY_LIMIT = 3
|
||||
|
||||
|
||||
def download_with_progressbar(url, save_path):
|
||||
logger = get_logger()
|
||||
if save_path and os.path.exists(save_path):
|
||||
logger.info(f"Path {save_path} already exists. Skipping...")
|
||||
return
|
||||
else:
|
||||
# Mainly used to solve the problem of downloading data from different
|
||||
# machines in the case of multiple machines. Different nodes will download
|
||||
# data, and the same node will only download data once.
|
||||
if dist.get_rank() == 0:
|
||||
_download(url, save_path)
|
||||
else:
|
||||
while not os.path.exists(save_path):
|
||||
time.sleep(1)
|
||||
|
||||
|
||||
def _download(url, save_path):
|
||||
"""
|
||||
Download from url, save to path.
|
||||
|
||||
url (str): download url
|
||||
save_path (str): download to given path
|
||||
"""
|
||||
logger = get_logger()
|
||||
|
||||
fname = osp.split(url)[-1]
|
||||
retry_cnt = 0
|
||||
|
||||
while not osp.exists(save_path):
|
||||
if retry_cnt < DOWNLOAD_RETRY_LIMIT:
|
||||
retry_cnt += 1
|
||||
else:
|
||||
raise RuntimeError(
|
||||
"Download from {} failed. " "Retry limit reached".format(url)
|
||||
)
|
||||
|
||||
try:
|
||||
req = requests.get(url, stream=True)
|
||||
except Exception as e: # requests.exceptions.ConnectionError
|
||||
logger.info(
|
||||
"Downloading {} from {} failed {} times with exception {}".format(
|
||||
fname, url, retry_cnt + 1, str(e)
|
||||
)
|
||||
)
|
||||
time.sleep(1)
|
||||
continue
|
||||
|
||||
if req.status_code != 200:
|
||||
raise RuntimeError(
|
||||
"Downloading from {} failed with code "
|
||||
"{}!".format(url, req.status_code)
|
||||
)
|
||||
|
||||
# For protecting download interrupted, download to
|
||||
# tmp_file firstly, move tmp_file to save_path
|
||||
# after download finished
|
||||
tmp_file = save_path + ".tmp"
|
||||
total_size = req.headers.get("content-length")
|
||||
with open(tmp_file, "wb") as f:
|
||||
if total_size:
|
||||
with tqdm(total=(int(total_size) + 1023) // 1024) as pbar:
|
||||
for chunk in req.iter_content(chunk_size=1024):
|
||||
f.write(chunk)
|
||||
pbar.update(1)
|
||||
else:
|
||||
for chunk in req.iter_content(chunk_size=1024):
|
||||
if chunk:
|
||||
f.write(chunk)
|
||||
shutil.move(tmp_file, save_path)
|
||||
|
||||
return save_path
|
||||
|
||||
|
||||
def maybe_download(model_storage_directory, url):
|
||||
# using custom model
|
||||
tar_file_name_list = [".pdiparams", ".pdiparams.info", ".pdmodel"]
|
||||
if not os.path.exists(
|
||||
os.path.join(model_storage_directory, "inference.pdiparams")
|
||||
) or not os.path.exists(os.path.join(model_storage_directory, "inference.pdmodel")):
|
||||
assert url.endswith(".tar"), "Only supports tar compressed package"
|
||||
tmp_path = os.path.join(model_storage_directory, url.split("/")[-1])
|
||||
print("download {} to {}".format(url, tmp_path))
|
||||
os.makedirs(model_storage_directory, exist_ok=True)
|
||||
download_with_progressbar(url, tmp_path)
|
||||
with tarfile.open(tmp_path, "r") as tarObj:
|
||||
for member in tarObj.getmembers():
|
||||
filename = None
|
||||
for tar_file_name in tar_file_name_list:
|
||||
if member.name.endswith(tar_file_name):
|
||||
filename = "inference" + tar_file_name
|
||||
if filename is None:
|
||||
continue
|
||||
file = tarObj.extractfile(member)
|
||||
with open(os.path.join(model_storage_directory, filename), "wb") as f:
|
||||
f.write(file.read())
|
||||
os.remove(tmp_path)
|
||||
|
||||
|
||||
def maybe_download_params(model_path):
|
||||
if os.path.exists(model_path) or not is_link(model_path):
|
||||
return model_path
|
||||
else:
|
||||
url = model_path
|
||||
tmp_path = os.path.join(MODELS_DIR, url.split("/")[-1])
|
||||
print("download {} to {}".format(url, tmp_path))
|
||||
os.makedirs(MODELS_DIR, exist_ok=True)
|
||||
download_with_progressbar(url, tmp_path)
|
||||
return tmp_path
|
||||
|
||||
|
||||
def is_link(s):
|
||||
return s is not None and s.startswith("http")
|
||||
|
||||
|
||||
def confirm_model_dir_url(model_dir, default_model_dir, default_url):
|
||||
url = default_url
|
||||
if model_dir is None or is_link(model_dir):
|
||||
if is_link(model_dir):
|
||||
url = model_dir
|
||||
file_name = url.split("/")[-1][:-4]
|
||||
model_dir = default_model_dir
|
||||
model_dir = os.path.join(model_dir, file_name)
|
||||
return model_dir, url
|
||||
146
ppocr/utils/poly_nms.py
Normal file
146
ppocr/utils/poly_nms.py
Normal file
@@ -0,0 +1,146 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import numpy as np
|
||||
from shapely.geometry import Polygon
|
||||
|
||||
|
||||
def points2polygon(points):
|
||||
"""Convert k points to 1 polygon.
|
||||
|
||||
Args:
|
||||
points (ndarray or list): A ndarray or a list of shape (2k)
|
||||
that indicates k points.
|
||||
|
||||
Returns:
|
||||
polygon (Polygon): A polygon object.
|
||||
"""
|
||||
if isinstance(points, list):
|
||||
points = np.array(points)
|
||||
|
||||
assert isinstance(points, np.ndarray)
|
||||
assert (points.size % 2 == 0) and (points.size >= 8)
|
||||
|
||||
point_mat = points.reshape([-1, 2])
|
||||
return Polygon(point_mat)
|
||||
|
||||
|
||||
def poly_intersection(poly_det, poly_gt, buffer=0.0001):
|
||||
"""Calculate the intersection area between two polygon.
|
||||
|
||||
Args:
|
||||
poly_det (Polygon): A polygon predicted by detector.
|
||||
poly_gt (Polygon): A gt polygon.
|
||||
|
||||
Returns:
|
||||
intersection_area (float): The intersection area between two polygons.
|
||||
"""
|
||||
assert isinstance(poly_det, Polygon)
|
||||
assert isinstance(poly_gt, Polygon)
|
||||
|
||||
if buffer == 0:
|
||||
poly_inter = poly_det & poly_gt
|
||||
else:
|
||||
poly_inter = poly_det.buffer(buffer) & poly_gt.buffer(buffer)
|
||||
return poly_inter.area, poly_inter
|
||||
|
||||
|
||||
def poly_union(poly_det, poly_gt):
|
||||
"""Calculate the union area between two polygon.
|
||||
|
||||
Args:
|
||||
poly_det (Polygon): A polygon predicted by detector.
|
||||
poly_gt (Polygon): A gt polygon.
|
||||
|
||||
Returns:
|
||||
union_area (float): The union area between two polygons.
|
||||
"""
|
||||
assert isinstance(poly_det, Polygon)
|
||||
assert isinstance(poly_gt, Polygon)
|
||||
|
||||
area_det = poly_det.area
|
||||
area_gt = poly_gt.area
|
||||
area_inters, _ = poly_intersection(poly_det, poly_gt)
|
||||
return area_det + area_gt - area_inters
|
||||
|
||||
|
||||
def valid_boundary(x, with_score=True):
|
||||
num = len(x)
|
||||
if num < 8:
|
||||
return False
|
||||
if num % 2 == 0 and (not with_score):
|
||||
return True
|
||||
if num % 2 == 1 and with_score:
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def boundary_iou(src, target):
|
||||
"""Calculate the IOU between two boundaries.
|
||||
|
||||
Args:
|
||||
src (list): Source boundary.
|
||||
target (list): Target boundary.
|
||||
|
||||
Returns:
|
||||
iou (float): The iou between two boundaries.
|
||||
"""
|
||||
assert valid_boundary(src, False)
|
||||
assert valid_boundary(target, False)
|
||||
src_poly = points2polygon(src)
|
||||
target_poly = points2polygon(target)
|
||||
|
||||
return poly_iou(src_poly, target_poly)
|
||||
|
||||
|
||||
def poly_iou(poly_det, poly_gt):
|
||||
"""Calculate the IOU between two polygons.
|
||||
|
||||
Args:
|
||||
poly_det (Polygon): A polygon predicted by detector.
|
||||
poly_gt (Polygon): A gt polygon.
|
||||
|
||||
Returns:
|
||||
iou (float): The IOU between two polygons.
|
||||
"""
|
||||
assert isinstance(poly_det, Polygon)
|
||||
assert isinstance(poly_gt, Polygon)
|
||||
area_inters, _ = poly_intersection(poly_det, poly_gt)
|
||||
area_union = poly_union(poly_det, poly_gt)
|
||||
if area_union == 0:
|
||||
return 0.0
|
||||
return area_inters / area_union
|
||||
|
||||
|
||||
def poly_nms(polygons, threshold):
|
||||
assert isinstance(polygons, list)
|
||||
|
||||
polygons = np.array(sorted(polygons, key=lambda x: x[-1]))
|
||||
|
||||
keep_poly = []
|
||||
index = [i for i in range(polygons.shape[0])]
|
||||
|
||||
while len(index) > 0:
|
||||
keep_poly.append(polygons[index[-1]].tolist())
|
||||
A = polygons[index[-1]][:-1]
|
||||
index = np.delete(index, -1)
|
||||
iou_list = np.zeros((len(index),))
|
||||
for i in range(len(index)):
|
||||
B = polygons[index[i]][:-1]
|
||||
iou_list[i] = boundary_iou(A, B)
|
||||
remove_index = np.where(iou_list > threshold)
|
||||
index = np.delete(index, remove_index)
|
||||
|
||||
return keep_poly
|
||||
6623
ppocr/utils/ppocr_keys_v1.txt
Normal file
6623
ppocr/utils/ppocr_keys_v1.txt
Normal file
File diff suppressed because it is too large
Load Diff
130
ppocr/utils/profiler.py
Normal file
130
ppocr/utils/profiler.py
Normal file
@@ -0,0 +1,130 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import sys
|
||||
import paddle.profiler as profiler
|
||||
|
||||
# A global variable to record the number of calling times for profiler
|
||||
# functions. It is used to specify the tracing range of training steps.
|
||||
_profiler_step_id = 0
|
||||
|
||||
# A global variable to avoid parsing from string every time.
|
||||
_profiler_options = None
|
||||
_prof = None
|
||||
|
||||
|
||||
class ProfilerOptions(object):
|
||||
"""
|
||||
Use a string to initialize a ProfilerOptions.
|
||||
The string should be in the format: "key1=value1;key2=value;key3=value3".
|
||||
For example:
|
||||
"profile_path=model.profile"
|
||||
"batch_range=[50, 60]; profile_path=model.profile"
|
||||
"batch_range=[50, 60]; tracer_option=OpDetail; profile_path=model.profile"
|
||||
|
||||
ProfilerOptions supports following key-value pair:
|
||||
batch_range - a integer list, e.g. [100, 110].
|
||||
state - a string, the optional values are 'CPU', 'GPU' or 'All'.
|
||||
sorted_key - a string, the optional values are 'calls', 'total',
|
||||
'max', 'min' or 'ave.
|
||||
tracer_option - a string, the optional values are 'Default', 'OpDetail',
|
||||
'AllOpDetail'.
|
||||
profile_path - a string, the path to save the serialized profile data,
|
||||
which can be used to generate a timeline.
|
||||
exit_on_finished - a boolean.
|
||||
"""
|
||||
|
||||
def __init__(self, options_str):
|
||||
assert isinstance(options_str, str)
|
||||
|
||||
self._options = {
|
||||
"batch_range": [10, 20],
|
||||
"state": "All",
|
||||
"sorted_key": "total",
|
||||
"tracer_option": "Default",
|
||||
"profile_path": "/tmp/profile",
|
||||
"exit_on_finished": True,
|
||||
"timer_only": True,
|
||||
}
|
||||
self._parse_from_string(options_str)
|
||||
|
||||
def _parse_from_string(self, options_str):
|
||||
for kv in options_str.replace(" ", "").split(";"):
|
||||
key, value = kv.split("=")
|
||||
if key == "batch_range":
|
||||
value_list = value.replace("[", "").replace("]", "").split(",")
|
||||
value_list = list(map(int, value_list))
|
||||
if (
|
||||
len(value_list) >= 2
|
||||
and value_list[0] >= 0
|
||||
and value_list[1] > value_list[0]
|
||||
):
|
||||
self._options[key] = value_list
|
||||
elif key == "exit_on_finished":
|
||||
self._options[key] = value.lower() in ("yes", "true", "t", "1")
|
||||
elif key in ["state", "sorted_key", "tracer_option", "profile_path"]:
|
||||
self._options[key] = value
|
||||
elif key == "timer_only":
|
||||
self._options[key] = value
|
||||
|
||||
def __getitem__(self, name):
|
||||
if self._options.get(name, None) is None:
|
||||
raise ValueError("ProfilerOptions does not have an option named %s." % name)
|
||||
return self._options[name]
|
||||
|
||||
|
||||
def add_profiler_step(options_str=None):
|
||||
"""
|
||||
Enable the operator-level timing using PaddlePaddle's profiler.
|
||||
The profiler uses a independent variable to count the profiler steps.
|
||||
One call of this function is treated as a profiler step.
|
||||
Args:
|
||||
profiler_options - a string to initialize the ProfilerOptions.
|
||||
Default is None, and the profiler is disabled.
|
||||
"""
|
||||
if options_str is None:
|
||||
return
|
||||
|
||||
global _prof
|
||||
global _profiler_step_id
|
||||
global _profiler_options
|
||||
|
||||
if _profiler_options is None:
|
||||
_profiler_options = ProfilerOptions(options_str)
|
||||
# profile : https://www.paddlepaddle.org.cn/documentation/docs/zh/guides/performance_improving/profiling_model.html#chakanxingnengshujudetongjibiaodan
|
||||
# timer_only = True only the model's throughput and time overhead are displayed
|
||||
# timer_only = False calling summary can print a statistical form that presents performance data from different perspectives.
|
||||
# timer_only = False the output Timeline information can be found in the profiler_log directory
|
||||
if _prof is None:
|
||||
_timer_only = str(_profiler_options["timer_only"]) == str(True)
|
||||
_prof = profiler.Profiler(
|
||||
scheduler=(
|
||||
_profiler_options["batch_range"][0],
|
||||
_profiler_options["batch_range"][1],
|
||||
),
|
||||
on_trace_ready=profiler.export_chrome_tracing("./profiler_log"),
|
||||
timer_only=_timer_only,
|
||||
)
|
||||
_prof.start()
|
||||
else:
|
||||
_prof.step()
|
||||
|
||||
if _profiler_step_id == _profiler_options["batch_range"][1]:
|
||||
_prof.stop()
|
||||
_prof.summary(op_detail=True, thread_sep=False, time_unit="ms")
|
||||
_prof = None
|
||||
if _profiler_options["exit_on_finished"]:
|
||||
sys.exit(0)
|
||||
|
||||
_profiler_step_id += 1
|
||||
390
ppocr/utils/save_load.py
Normal file
390
ppocr/utils/save_load.py
Normal file
@@ -0,0 +1,390 @@
|
||||
# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import errno
|
||||
import os
|
||||
import pickle
|
||||
import json
|
||||
from packaging import version
|
||||
|
||||
import paddle
|
||||
|
||||
from ppocr.utils.logging import get_logger
|
||||
from ppocr.utils.network import maybe_download_params
|
||||
|
||||
try:
|
||||
import encryption # Attempt to import the encryption module for AIStudio's encryption model
|
||||
|
||||
encrypted = encryption.is_encryption_needed()
|
||||
except ImportError:
|
||||
get_logger().warning("Skipping import of the encryption module.")
|
||||
encrypted = False # Encryption is not needed if the module cannot be imported
|
||||
|
||||
__all__ = ["load_model"]
|
||||
|
||||
|
||||
# just to determine the inference model file format
|
||||
def get_FLAGS_json_format_model():
|
||||
# json format by default
|
||||
return os.environ.get("FLAGS_json_format_model", "1").lower() in ("1", "true", "t")
|
||||
|
||||
|
||||
FLAGS_json_format_model = get_FLAGS_json_format_model()
|
||||
|
||||
|
||||
def _mkdir_if_not_exist(path, logger):
|
||||
"""
|
||||
mkdir if not exists, ignore the exception when multiprocess mkdir together
|
||||
"""
|
||||
if not os.path.exists(path):
|
||||
try:
|
||||
os.makedirs(path)
|
||||
except OSError as e:
|
||||
if e.errno == errno.EEXIST and os.path.isdir(path):
|
||||
logger.warning(
|
||||
"be happy if some process has already created {}".format(path)
|
||||
)
|
||||
else:
|
||||
raise OSError("Failed to mkdir {}".format(path))
|
||||
|
||||
|
||||
def load_model(config, model, optimizer=None, model_type="det"):
|
||||
"""
|
||||
load model from checkpoint or pretrained_model
|
||||
"""
|
||||
logger = get_logger()
|
||||
global_config = config["Global"]
|
||||
checkpoints = global_config.get("checkpoints")
|
||||
pretrained_model = global_config.get("pretrained_model")
|
||||
best_model_dict = {}
|
||||
is_float16 = False
|
||||
is_nlp_model = model_type == "kie" and config["Architecture"]["algorithm"] not in [
|
||||
"SDMGR"
|
||||
]
|
||||
|
||||
if is_nlp_model is True:
|
||||
# NOTE: for kie model dsitillation, resume training is not supported now
|
||||
if config["Architecture"]["algorithm"] in ["Distillation"]:
|
||||
return best_model_dict
|
||||
checkpoints = config["Architecture"]["Backbone"]["checkpoints"]
|
||||
# load kie method metric
|
||||
if checkpoints:
|
||||
if os.path.exists(os.path.join(checkpoints, "metric.states")):
|
||||
with open(os.path.join(checkpoints, "metric.states"), "rb") as f:
|
||||
states_dict = pickle.load(f, encoding="latin1")
|
||||
best_model_dict = states_dict.get("best_model_dict", {})
|
||||
if "epoch" in states_dict:
|
||||
best_model_dict["start_epoch"] = states_dict["epoch"] + 1
|
||||
logger.info("resume from {}".format(checkpoints))
|
||||
|
||||
if optimizer is not None:
|
||||
if checkpoints[-1] in ["/", "\\"]:
|
||||
checkpoints = checkpoints[:-1]
|
||||
if os.path.exists(checkpoints + ".pdopt"):
|
||||
optim_dict = paddle.load(checkpoints + ".pdopt")
|
||||
optimizer.set_state_dict(optim_dict)
|
||||
else:
|
||||
logger.warning(
|
||||
"{}.pdopt is not exists, params of optimizer is not loaded".format(
|
||||
checkpoints
|
||||
)
|
||||
)
|
||||
|
||||
return best_model_dict
|
||||
|
||||
if checkpoints:
|
||||
if checkpoints.endswith(".pdparams"):
|
||||
checkpoints = checkpoints.replace(".pdparams", "")
|
||||
assert os.path.exists(
|
||||
checkpoints + ".pdparams"
|
||||
), "The {}.pdparams does not exists!".format(checkpoints)
|
||||
|
||||
# load params from trained model
|
||||
params = paddle.load(checkpoints + ".pdparams")
|
||||
state_dict = model.state_dict()
|
||||
new_state_dict = {}
|
||||
for key, value in state_dict.items():
|
||||
if key not in params:
|
||||
logger.warning(
|
||||
"{} not in loaded params {} !".format(key, params.keys())
|
||||
)
|
||||
continue
|
||||
pre_value = params[key]
|
||||
if pre_value.dtype == paddle.float16:
|
||||
is_float16 = True
|
||||
if pre_value.dtype != value.dtype:
|
||||
pre_value = pre_value.astype(value.dtype)
|
||||
if list(value.shape) == list(pre_value.shape):
|
||||
new_state_dict[key] = pre_value
|
||||
else:
|
||||
logger.warning(
|
||||
"The shape of model params {} {} not matched with loaded params shape {} !".format(
|
||||
key, value.shape, pre_value.shape
|
||||
)
|
||||
)
|
||||
model.set_state_dict(new_state_dict)
|
||||
if is_float16:
|
||||
logger.info(
|
||||
"The parameter type is float16, which is converted to float32 when loading"
|
||||
)
|
||||
if optimizer is not None:
|
||||
if os.path.exists(checkpoints + ".pdopt"):
|
||||
optim_dict = paddle.load(checkpoints + ".pdopt")
|
||||
optimizer.set_state_dict(optim_dict)
|
||||
else:
|
||||
logger.warning(
|
||||
"{}.pdopt is not exists, params of optimizer is not loaded".format(
|
||||
checkpoints
|
||||
)
|
||||
)
|
||||
|
||||
if os.path.exists(checkpoints + ".states"):
|
||||
with open(checkpoints + ".states", "rb") as f:
|
||||
states_dict = pickle.load(f, encoding="latin1")
|
||||
best_model_dict = states_dict.get("best_model_dict", {})
|
||||
best_model_dict["acc"] = 0.0
|
||||
if "epoch" in states_dict:
|
||||
best_model_dict["start_epoch"] = states_dict["epoch"] + 1
|
||||
logger.info("resume from {}".format(checkpoints))
|
||||
elif pretrained_model:
|
||||
is_float16 = load_pretrained_params(model, pretrained_model)
|
||||
else:
|
||||
logger.info("train from scratch")
|
||||
best_model_dict["is_float16"] = is_float16
|
||||
return best_model_dict
|
||||
|
||||
|
||||
def load_pretrained_params(model, path):
|
||||
logger = get_logger()
|
||||
path = maybe_download_params(path)
|
||||
if path.endswith(".pdparams"):
|
||||
path = path.replace(".pdparams", "")
|
||||
assert os.path.exists(
|
||||
path + ".pdparams"
|
||||
), "The {}.pdparams does not exists!".format(path)
|
||||
|
||||
params = paddle.load(path + ".pdparams")
|
||||
|
||||
state_dict = model.state_dict()
|
||||
|
||||
new_state_dict = {}
|
||||
is_float16 = False
|
||||
|
||||
for k1 in params.keys():
|
||||
if k1 not in state_dict.keys():
|
||||
logger.warning("The pretrained params {} not in model".format(k1))
|
||||
else:
|
||||
if params[k1].dtype == paddle.float16:
|
||||
is_float16 = True
|
||||
if params[k1].dtype != state_dict[k1].dtype:
|
||||
params[k1] = params[k1].astype(state_dict[k1].dtype)
|
||||
if list(state_dict[k1].shape) == list(params[k1].shape):
|
||||
new_state_dict[k1] = params[k1]
|
||||
else:
|
||||
logger.warning(
|
||||
"The shape of model params {} {} not matched with loaded params {} {} !".format(
|
||||
k1, state_dict[k1].shape, k1, params[k1].shape
|
||||
)
|
||||
)
|
||||
|
||||
model.set_state_dict(new_state_dict)
|
||||
if is_float16:
|
||||
logger.info(
|
||||
"The parameter type is float16, which is converted to float32 when loading"
|
||||
)
|
||||
logger.info("load pretrain successful from {}".format(path))
|
||||
return is_float16
|
||||
|
||||
|
||||
def save_model(
|
||||
model,
|
||||
optimizer,
|
||||
model_path,
|
||||
logger,
|
||||
config,
|
||||
is_best=False,
|
||||
prefix="ppocr",
|
||||
**kwargs,
|
||||
):
|
||||
"""
|
||||
save model to the target path
|
||||
"""
|
||||
_mkdir_if_not_exist(model_path, logger)
|
||||
model_prefix = os.path.join(model_path, prefix)
|
||||
|
||||
if prefix == "best_accuracy":
|
||||
best_model_path = os.path.join(model_path, "best_model")
|
||||
_mkdir_if_not_exist(best_model_path, logger)
|
||||
|
||||
paddle.save(optimizer.state_dict(), model_prefix + ".pdopt")
|
||||
if prefix == "best_accuracy":
|
||||
paddle.save(
|
||||
optimizer.state_dict(), os.path.join(best_model_path, "model.pdopt")
|
||||
)
|
||||
|
||||
is_nlp_model = config["Architecture"]["model_type"] == "kie" and config[
|
||||
"Architecture"
|
||||
]["algorithm"] not in ["SDMGR"]
|
||||
if is_nlp_model is not True:
|
||||
paddle.save(model.state_dict(), model_prefix + ".pdparams")
|
||||
metric_prefix = model_prefix
|
||||
|
||||
if prefix == "best_accuracy":
|
||||
paddle.save(
|
||||
model.state_dict(), os.path.join(best_model_path, "model.pdparams")
|
||||
)
|
||||
|
||||
else: # for kie system, we follow the save/load rules in NLP
|
||||
if config["Global"]["distributed"]:
|
||||
arch = model._layers
|
||||
else:
|
||||
arch = model
|
||||
if config["Architecture"]["algorithm"] in ["Distillation"]:
|
||||
arch = arch.Student
|
||||
arch.backbone.model.save_pretrained(model_prefix)
|
||||
metric_prefix = os.path.join(model_prefix, "metric")
|
||||
|
||||
if prefix == "best_accuracy":
|
||||
arch.backbone.model.save_pretrained(best_model_path)
|
||||
|
||||
save_model_info = kwargs.pop("save_model_info", False)
|
||||
if save_model_info:
|
||||
with open(os.path.join(model_path, f"{prefix}.info.json"), "w") as f:
|
||||
json.dump(kwargs, f)
|
||||
logger.info("Already save model info in {}".format(model_path))
|
||||
if prefix != "latest":
|
||||
done_flag = kwargs.pop("done_flag", False)
|
||||
update_train_results(config, prefix, save_model_info, done_flag=done_flag)
|
||||
|
||||
# save metric and config
|
||||
with open(metric_prefix + ".states", "wb") as f:
|
||||
pickle.dump(kwargs, f, protocol=2)
|
||||
if is_best:
|
||||
logger.info("save best model is to {}".format(model_prefix))
|
||||
else:
|
||||
logger.info("save model in {}".format(model_prefix))
|
||||
|
||||
|
||||
def update_train_results(config, prefix, metric_info, done_flag=False, last_num=5):
|
||||
if paddle.distributed.get_rank() != 0:
|
||||
return
|
||||
|
||||
assert last_num >= 1
|
||||
train_results_path = os.path.join(
|
||||
config["Global"]["save_model_dir"], "train_result.json"
|
||||
)
|
||||
save_model_tag = ["pdparams", "pdopt", "pdstates"]
|
||||
paddle_version = version.parse(paddle.__version__)
|
||||
if FLAGS_json_format_model or paddle_version >= version.parse("3.0.0"):
|
||||
save_inference_files = {
|
||||
"inference_config": "inference.yml",
|
||||
"pdmodel": "inference.json",
|
||||
"pdiparams": "inference.pdiparams",
|
||||
}
|
||||
else:
|
||||
save_inference_files = {
|
||||
"inference_config": "inference.yml",
|
||||
"pdmodel": "inference.pdmodel",
|
||||
"pdiparams": "inference.pdiparams",
|
||||
"pdiparams.info": "inference.pdiparams.info",
|
||||
}
|
||||
if os.path.exists(train_results_path):
|
||||
with open(train_results_path, "r") as fp:
|
||||
train_results = json.load(fp)
|
||||
else:
|
||||
train_results = {}
|
||||
train_results["model_name"] = config["Global"]["model_name"]
|
||||
label_dict_path = config["Global"].get("character_dict_path", "")
|
||||
if label_dict_path != "":
|
||||
label_dict_path = os.path.abspath(label_dict_path)
|
||||
if not os.path.exists(label_dict_path):
|
||||
label_dict_path = ""
|
||||
train_results["label_dict"] = label_dict_path
|
||||
train_results["train_log"] = "train.log"
|
||||
train_results["visualdl_log"] = ""
|
||||
train_results["config"] = "config.yaml"
|
||||
train_results["models"] = {}
|
||||
for i in range(1, last_num + 1):
|
||||
train_results["models"][f"last_{i}"] = {}
|
||||
train_results["models"]["best"] = {}
|
||||
train_results["done_flag"] = done_flag
|
||||
if "best" in prefix:
|
||||
if "acc" in metric_info["metric"]:
|
||||
metric_score = metric_info["metric"]["acc"]
|
||||
elif "precision" in metric_info["metric"]:
|
||||
metric_score = metric_info["metric"]["precision"]
|
||||
elif "exp_rate" in metric_info["metric"]:
|
||||
metric_score = metric_info["metric"]["exp_rate"]
|
||||
else:
|
||||
raise ValueError("No metric score found.")
|
||||
train_results["models"]["best"]["score"] = metric_score
|
||||
for tag in save_model_tag:
|
||||
if tag == "pdparams" and encrypted:
|
||||
train_results["models"]["best"][tag] = os.path.join(
|
||||
prefix,
|
||||
(
|
||||
f"{prefix}.encrypted.{tag}"
|
||||
if tag != "pdstates"
|
||||
else f"{prefix}.states"
|
||||
),
|
||||
)
|
||||
else:
|
||||
train_results["models"]["best"][tag] = os.path.join(
|
||||
prefix,
|
||||
f"{prefix}.{tag}" if tag != "pdstates" else f"{prefix}.states",
|
||||
)
|
||||
for key in save_inference_files:
|
||||
train_results["models"]["best"][key] = os.path.join(
|
||||
prefix, "inference", save_inference_files[key]
|
||||
)
|
||||
else:
|
||||
for i in range(last_num - 1, 0, -1):
|
||||
train_results["models"][f"last_{i + 1}"] = train_results["models"][
|
||||
f"last_{i}"
|
||||
].copy()
|
||||
if "acc" in metric_info["metric"]:
|
||||
metric_score = metric_info["metric"]["acc"]
|
||||
elif "precision" in metric_info["metric"]:
|
||||
metric_score = metric_info["metric"]["precision"]
|
||||
elif "exp_rate" in metric_info["metric"]:
|
||||
metric_score = metric_info["metric"]["exp_rate"]
|
||||
else:
|
||||
metric_score = 0
|
||||
train_results["models"][f"last_{1}"]["score"] = metric_score
|
||||
for tag in save_model_tag:
|
||||
if tag == "pdparams" and encrypted:
|
||||
train_results["models"][f"last_{1}"][tag] = os.path.join(
|
||||
prefix,
|
||||
(
|
||||
f"{prefix}.encrypted.{tag}"
|
||||
if tag != "pdstates"
|
||||
else f"{prefix}.states"
|
||||
),
|
||||
)
|
||||
else:
|
||||
train_results["models"][f"last_{1}"][tag] = os.path.join(
|
||||
prefix,
|
||||
f"{prefix}.{tag}" if tag != "pdstates" else f"{prefix}.states",
|
||||
)
|
||||
for key in save_inference_files:
|
||||
train_results["models"][f"last_{1}"][key] = os.path.join(
|
||||
prefix, "inference", save_inference_files[key]
|
||||
)
|
||||
|
||||
with open(train_results_path, "w") as fp:
|
||||
json.dump(train_results, fp)
|
||||
70
ppocr/utils/stats.py
Executable file
70
ppocr/utils/stats.py
Executable file
@@ -0,0 +1,70 @@
|
||||
# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import collections
|
||||
import numpy as np
|
||||
import datetime
|
||||
|
||||
__all__ = ["TrainingStats", "Time"]
|
||||
|
||||
|
||||
class SmoothedValue(object):
|
||||
"""Track a series of values and provide access to smoothed values over a
|
||||
window or the global series average.
|
||||
"""
|
||||
|
||||
def __init__(self, window_size):
|
||||
self.deque = collections.deque(maxlen=window_size)
|
||||
|
||||
def add_value(self, value):
|
||||
self.deque.append(value)
|
||||
|
||||
def get_median_value(self):
|
||||
return np.median(self.deque)
|
||||
|
||||
|
||||
def Time():
|
||||
return datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S.%f")
|
||||
|
||||
|
||||
class TrainingStats(object):
|
||||
def __init__(self, window_size, stats_keys):
|
||||
self.window_size = window_size
|
||||
self.smoothed_losses_and_metrics = {
|
||||
key: SmoothedValue(window_size) for key in stats_keys
|
||||
}
|
||||
|
||||
def update(self, stats):
|
||||
for k, v in stats.items():
|
||||
if k not in self.smoothed_losses_and_metrics:
|
||||
self.smoothed_losses_and_metrics[k] = SmoothedValue(self.window_size)
|
||||
self.smoothed_losses_and_metrics[k].add_value(v)
|
||||
|
||||
def get(self, extras=None):
|
||||
stats = collections.OrderedDict()
|
||||
if extras:
|
||||
for k, v in extras.items():
|
||||
stats[k] = v
|
||||
for k, v in self.smoothed_losses_and_metrics.items():
|
||||
stats[k] = round(v.get_median_value(), 6)
|
||||
|
||||
return stats
|
||||
|
||||
def log(self, extras=None):
|
||||
d = self.get(extras)
|
||||
strs = []
|
||||
for k, v in d.items():
|
||||
strs.append("{}: {:x<6f}".format(k, v))
|
||||
strs = ", ".join(strs)
|
||||
return strs
|
||||
216
ppocr/utils/utility.py
Normal file
216
ppocr/utils/utility.py
Normal file
@@ -0,0 +1,216 @@
|
||||
# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import logging
|
||||
import os
|
||||
import cv2
|
||||
import random
|
||||
import numpy as np
|
||||
import paddle
|
||||
import importlib.util
|
||||
import sys
|
||||
import subprocess
|
||||
|
||||
|
||||
def print_dict(d, logger, delimiter=0):
|
||||
"""
|
||||
Recursively visualize a dict and
|
||||
indenting acrrording by the relationship of keys.
|
||||
"""
|
||||
for k, v in sorted(d.items()):
|
||||
if isinstance(v, dict):
|
||||
logger.info("{}{} : ".format(delimiter * " ", str(k)))
|
||||
print_dict(v, logger, delimiter + 4)
|
||||
elif isinstance(v, list) and len(v) >= 1 and isinstance(v[0], dict):
|
||||
logger.info("{}{} : ".format(delimiter * " ", str(k)))
|
||||
for value in v:
|
||||
print_dict(value, logger, delimiter + 4)
|
||||
else:
|
||||
logger.info("{}{} : {}".format(delimiter * " ", k, v))
|
||||
|
||||
|
||||
def get_check_global_params(mode):
|
||||
check_params = [
|
||||
"use_gpu",
|
||||
"max_text_length",
|
||||
"image_shape",
|
||||
"image_shape",
|
||||
"character_type",
|
||||
"loss_type",
|
||||
]
|
||||
if mode == "train_eval":
|
||||
check_params = check_params + [
|
||||
"train_batch_size_per_card",
|
||||
"test_batch_size_per_card",
|
||||
]
|
||||
elif mode == "test":
|
||||
check_params = check_params + ["test_batch_size_per_card"]
|
||||
return check_params
|
||||
|
||||
|
||||
def _check_image_file(path):
|
||||
img_end = {"jpg", "bmp", "png", "jpeg", "rgb", "tif", "tiff", "gif", "pdf"}
|
||||
return any([path.lower().endswith(e) for e in img_end])
|
||||
|
||||
|
||||
def get_image_file_list(img_file, infer_list=None):
|
||||
imgs_lists = []
|
||||
if infer_list and not os.path.exists(infer_list):
|
||||
raise Exception("not found infer list {}".format(infer_list))
|
||||
if infer_list:
|
||||
with open(infer_list, "r") as f:
|
||||
lines = f.readlines()
|
||||
for line in lines:
|
||||
image_path = line.strip().split("\t")[0]
|
||||
image_path = os.path.join(img_file, image_path)
|
||||
imgs_lists.append(image_path)
|
||||
else:
|
||||
if img_file is None or not os.path.exists(img_file):
|
||||
raise Exception("not found any img file in {}".format(img_file))
|
||||
|
||||
img_end = {"jpg", "bmp", "png", "jpeg", "rgb", "tif", "tiff", "gif", "pdf"}
|
||||
if os.path.isfile(img_file) and _check_image_file(img_file):
|
||||
imgs_lists.append(img_file)
|
||||
elif os.path.isdir(img_file):
|
||||
for single_file in os.listdir(img_file):
|
||||
file_path = os.path.join(img_file, single_file)
|
||||
if os.path.isfile(file_path) and _check_image_file(file_path):
|
||||
imgs_lists.append(file_path)
|
||||
|
||||
if len(imgs_lists) == 0:
|
||||
raise Exception("not found any img file in {}".format(img_file))
|
||||
imgs_lists = sorted(imgs_lists)
|
||||
return imgs_lists
|
||||
|
||||
|
||||
def binarize_img(img):
|
||||
if len(img.shape) == 3 and img.shape[2] == 3:
|
||||
gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) # conversion to grayscale image
|
||||
# use cv2 threshold binarization
|
||||
_, gray = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU)
|
||||
img = cv2.cvtColor(gray, cv2.COLOR_GRAY2BGR)
|
||||
return img
|
||||
|
||||
|
||||
def alpha_to_color(img, alpha_color=(255, 255, 255)):
|
||||
if len(img.shape) == 3 and img.shape[2] == 4:
|
||||
B, G, R, A = cv2.split(img)
|
||||
alpha = A / 255
|
||||
|
||||
R = (alpha_color[0] * (1 - alpha) + R * alpha).astype(np.uint8)
|
||||
G = (alpha_color[1] * (1 - alpha) + G * alpha).astype(np.uint8)
|
||||
B = (alpha_color[2] * (1 - alpha) + B * alpha).astype(np.uint8)
|
||||
|
||||
img = cv2.merge((B, G, R))
|
||||
return img
|
||||
|
||||
|
||||
def check_and_read(img_path):
|
||||
if os.path.basename(img_path)[-3:].lower() == "gif":
|
||||
gif = cv2.VideoCapture(img_path)
|
||||
ret, frame = gif.read()
|
||||
if not ret:
|
||||
logger = logging.getLogger("ppocr")
|
||||
logger.info("Cannot read {}. This gif image maybe corrupted.")
|
||||
return None, False
|
||||
if len(frame.shape) == 2 or frame.shape[-1] == 1:
|
||||
frame = cv2.cvtColor(frame, cv2.COLOR_GRAY2RGB)
|
||||
imgvalue = frame[:, :, ::-1]
|
||||
return imgvalue, True, False
|
||||
elif os.path.basename(img_path)[-3:].lower() == "pdf":
|
||||
from paddle.utils import try_import
|
||||
|
||||
fitz = try_import("fitz")
|
||||
from PIL import Image
|
||||
|
||||
imgs = []
|
||||
with fitz.open(img_path) as pdf:
|
||||
for pg in range(0, pdf.page_count):
|
||||
page = pdf[pg]
|
||||
mat = fitz.Matrix(2, 2)
|
||||
pm = page.get_pixmap(matrix=mat, alpha=False)
|
||||
|
||||
# if width or height > 2000 pixels, don't enlarge the image
|
||||
if pm.width > 2000 or pm.height > 2000:
|
||||
pm = page.get_pixmap(matrix=fitz.Matrix(1, 1), alpha=False)
|
||||
|
||||
img = Image.frombytes("RGB", [pm.width, pm.height], pm.samples)
|
||||
img = cv2.cvtColor(np.array(img), cv2.COLOR_RGB2BGR)
|
||||
imgs.append(img)
|
||||
return imgs, False, True
|
||||
return None, False, False
|
||||
|
||||
|
||||
def load_vqa_bio_label_maps(label_map_path):
|
||||
with open(label_map_path, "r", encoding="utf-8") as fin:
|
||||
lines = fin.readlines()
|
||||
old_lines = [line.strip() for line in lines]
|
||||
lines = ["O"]
|
||||
for line in old_lines:
|
||||
# "O" has already been in lines
|
||||
if line.upper() in ["OTHER", "OTHERS", "IGNORE"]:
|
||||
continue
|
||||
lines.append(line)
|
||||
labels = ["O"]
|
||||
for line in lines[1:]:
|
||||
labels.append("B-" + line)
|
||||
labels.append("I-" + line)
|
||||
label2id_map = {label.upper(): idx for idx, label in enumerate(labels)}
|
||||
id2label_map = {idx: label.upper() for idx, label in enumerate(labels)}
|
||||
return label2id_map, id2label_map
|
||||
|
||||
|
||||
def set_seed(seed=1024):
|
||||
random.seed(seed)
|
||||
np.random.seed(seed)
|
||||
paddle.seed(seed)
|
||||
|
||||
|
||||
def check_install(module_name, install_name):
|
||||
spec = importlib.util.find_spec(module_name)
|
||||
if spec is None:
|
||||
print(f"Warning! The {module_name} module is NOT installed")
|
||||
print(
|
||||
f"Try install {module_name} module automatically. You can also try to install manually by pip install {install_name}."
|
||||
)
|
||||
python = sys.executable
|
||||
try:
|
||||
subprocess.check_call(
|
||||
[python, "-m", "pip", "install", install_name],
|
||||
stdout=subprocess.DEVNULL,
|
||||
)
|
||||
print(f"The {module_name} module is now installed")
|
||||
except subprocess.CalledProcessError as exc:
|
||||
raise Exception(f"Install {module_name} failed, please install manually")
|
||||
else:
|
||||
print(f"{module_name} has been installed.")
|
||||
|
||||
|
||||
class AverageMeter:
|
||||
def __init__(self):
|
||||
self.reset()
|
||||
|
||||
def reset(self):
|
||||
"""reset"""
|
||||
self.val = 0
|
||||
self.avg = 0
|
||||
self.sum = 0
|
||||
self.count = 0
|
||||
|
||||
def update(self, val, n=1):
|
||||
"""update"""
|
||||
self.val = val
|
||||
self.sum += val * n
|
||||
self.count += n
|
||||
self.avg = self.sum / self.count
|
||||
142
ppocr/utils/visual.py
Normal file
142
ppocr/utils/visual.py
Normal file
@@ -0,0 +1,142 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
import cv2
|
||||
import os
|
||||
import numpy as np
|
||||
import PIL
|
||||
from PIL import Image, ImageDraw, ImageFont
|
||||
|
||||
|
||||
def draw_ser_results(
|
||||
image, ocr_results, font_path="doc/fonts/simfang.ttf", font_size=14
|
||||
):
|
||||
np.random.seed(2021)
|
||||
color = (
|
||||
np.random.permutation(range(255)),
|
||||
np.random.permutation(range(255)),
|
||||
np.random.permutation(range(255)),
|
||||
)
|
||||
color_map = {
|
||||
idx: (color[0][idx], color[1][idx], color[2][idx]) for idx in range(1, 255)
|
||||
}
|
||||
if isinstance(image, np.ndarray):
|
||||
image = Image.fromarray(image)
|
||||
elif isinstance(image, str) and os.path.isfile(image):
|
||||
image = Image.open(image).convert("RGB")
|
||||
img_new = image.copy()
|
||||
draw = ImageDraw.Draw(img_new)
|
||||
|
||||
font = ImageFont.truetype(font_path, font_size, encoding="utf-8")
|
||||
for ocr_info in ocr_results:
|
||||
if ocr_info["pred_id"] not in color_map:
|
||||
continue
|
||||
color = color_map[ocr_info["pred_id"]]
|
||||
text = "{}: {}".format(ocr_info["pred"], ocr_info["transcription"])
|
||||
|
||||
if "bbox" in ocr_info:
|
||||
# draw with ocr engine
|
||||
bbox = ocr_info["bbox"]
|
||||
else:
|
||||
# draw with ocr groundtruth
|
||||
bbox = trans_poly_to_bbox(ocr_info["points"])
|
||||
draw_box_txt(bbox, text, draw, font, font_size, color)
|
||||
|
||||
img_new = Image.blend(image, img_new, 0.7)
|
||||
return np.array(img_new)
|
||||
|
||||
|
||||
def draw_box_txt(bbox, text, draw, font, font_size, color):
|
||||
# draw ocr results outline
|
||||
bbox = ((bbox[0], bbox[1]), (bbox[2], bbox[3]))
|
||||
draw.rectangle(bbox, fill=color)
|
||||
|
||||
# draw ocr results
|
||||
if int(PIL.__version__.split(".")[0]) < 10:
|
||||
tw = font.getsize(text)[0]
|
||||
th = font.getsize(text)[1]
|
||||
else:
|
||||
left, top, right, bottom = font.getbbox(text)
|
||||
tw, th = right - left, bottom - top
|
||||
|
||||
start_y = max(0, bbox[0][1] - th)
|
||||
draw.rectangle(
|
||||
[(bbox[0][0] + 1, start_y), (bbox[0][0] + tw + 1, start_y + th)],
|
||||
fill=(0, 0, 255),
|
||||
)
|
||||
draw.text((bbox[0][0] + 1, start_y), text, fill=(255, 255, 255), font=font)
|
||||
|
||||
|
||||
def trans_poly_to_bbox(poly):
|
||||
x1 = np.min([p[0] for p in poly])
|
||||
x2 = np.max([p[0] for p in poly])
|
||||
y1 = np.min([p[1] for p in poly])
|
||||
y2 = np.max([p[1] for p in poly])
|
||||
return [x1, y1, x2, y2]
|
||||
|
||||
|
||||
def draw_re_results(image, result, font_path="doc/fonts/simfang.ttf", font_size=18):
|
||||
np.random.seed(0)
|
||||
if isinstance(image, np.ndarray):
|
||||
image = Image.fromarray(image)
|
||||
elif isinstance(image, str) and os.path.isfile(image):
|
||||
image = Image.open(image).convert("RGB")
|
||||
img_new = image.copy()
|
||||
draw = ImageDraw.Draw(img_new)
|
||||
|
||||
font = ImageFont.truetype(font_path, font_size, encoding="utf-8")
|
||||
color_head = (0, 0, 255)
|
||||
color_tail = (255, 0, 0)
|
||||
color_line = (0, 255, 0)
|
||||
|
||||
for ocr_info_head, ocr_info_tail in result:
|
||||
draw_box_txt(
|
||||
ocr_info_head["bbox"],
|
||||
ocr_info_head["transcription"],
|
||||
draw,
|
||||
font,
|
||||
font_size,
|
||||
color_head,
|
||||
)
|
||||
draw_box_txt(
|
||||
ocr_info_tail["bbox"],
|
||||
ocr_info_tail["transcription"],
|
||||
draw,
|
||||
font,
|
||||
font_size,
|
||||
color_tail,
|
||||
)
|
||||
|
||||
center_head = (
|
||||
(ocr_info_head["bbox"][0] + ocr_info_head["bbox"][2]) // 2,
|
||||
(ocr_info_head["bbox"][1] + ocr_info_head["bbox"][3]) // 2,
|
||||
)
|
||||
center_tail = (
|
||||
(ocr_info_tail["bbox"][0] + ocr_info_tail["bbox"][2]) // 2,
|
||||
(ocr_info_tail["bbox"][1] + ocr_info_tail["bbox"][3]) // 2,
|
||||
)
|
||||
|
||||
draw.line([center_head, center_tail], fill=color_line, width=5)
|
||||
|
||||
img_new = Image.blend(image, img_new, 0.5)
|
||||
return np.array(img_new)
|
||||
|
||||
|
||||
def draw_rectangle(img_path, boxes):
|
||||
boxes = np.array(boxes)
|
||||
img = cv2.imread(img_path)
|
||||
img_show = img.copy()
|
||||
for box in boxes.astype(int):
|
||||
x1, y1, x2, y2 = box
|
||||
cv2.rectangle(img_show, (x1, y1), (x2, y2), (255, 0, 0), 2)
|
||||
return img_show
|
||||
Reference in New Issue
Block a user