This commit is contained in:
152
ppocr/modeling/backbones/__init__.py
Executable file
152
ppocr/modeling/backbones/__init__.py
Executable file
@@ -0,0 +1,152 @@
|
||||
# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
__all__ = ["build_backbone"]
|
||||
|
||||
|
||||
def build_backbone(config, model_type):
|
||||
if model_type == "det" or model_type == "table":
|
||||
from .det_mobilenet_v3 import MobileNetV3
|
||||
from .det_resnet import ResNet
|
||||
from .det_resnet_vd import ResNet_vd
|
||||
from .det_resnet_vd_sast import ResNet_SAST
|
||||
from .det_pp_lcnet import PPLCNet
|
||||
from .rec_lcnetv3 import PPLCNetV3
|
||||
from .rec_hgnet import PPHGNet_small
|
||||
from .rec_vit import ViT
|
||||
from .det_pp_lcnet_v2 import PPLCNetV2_base
|
||||
from .rec_repvit import RepSVTR_det
|
||||
from .rec_vary_vit import Vary_VIT_B
|
||||
from .rec_pphgnetv2 import PPHGNetV2_B4
|
||||
|
||||
support_dict = [
|
||||
"MobileNetV3",
|
||||
"ResNet",
|
||||
"ResNet_vd",
|
||||
"ResNet_SAST",
|
||||
"PPLCNet",
|
||||
"PPLCNetV3",
|
||||
"PPHGNet_small",
|
||||
"PPLCNetV2_base",
|
||||
"RepSVTR_det",
|
||||
"Vary_VIT_B",
|
||||
"PPHGNetV2_B4",
|
||||
]
|
||||
if model_type == "table":
|
||||
from .table_master_resnet import TableResNetExtra
|
||||
|
||||
support_dict.append("TableResNetExtra")
|
||||
elif model_type == "rec" or model_type == "cls":
|
||||
from .rec_mobilenet_v3 import MobileNetV3
|
||||
from .rec_resnet_vd import ResNet
|
||||
from .rec_resnet_fpn import ResNetFPN
|
||||
from .rec_mv1_enhance import MobileNetV1Enhance
|
||||
from .rec_nrtr_mtb import MTB
|
||||
from .rec_resnet_31 import ResNet31
|
||||
from .rec_resnet_32 import ResNet32
|
||||
from .rec_resnet_45 import ResNet45
|
||||
from .rec_resnet_aster import ResNet_ASTER
|
||||
from .rec_micronet import MicroNet
|
||||
from .rec_efficientb3_pren import EfficientNetb3_PREN
|
||||
from .rec_svtrnet import SVTRNet
|
||||
from .rec_vitstr import ViTSTR
|
||||
from .rec_resnet_rfl import ResNetRFL
|
||||
from .rec_densenet import DenseNet
|
||||
from .rec_resnetv2 import ResNetV2
|
||||
from .rec_hybridvit import HybridTransformer
|
||||
from .rec_donut_swin import DonutSwinModel
|
||||
from .rec_shallow_cnn import ShallowCNN
|
||||
from .rec_lcnetv3 import PPLCNetV3
|
||||
from .rec_hgnet import PPHGNet_small
|
||||
from .rec_vit_parseq import ViTParseQ
|
||||
from .rec_repvit import RepSVTR
|
||||
from .rec_svtrv2 import SVTRv2
|
||||
from .rec_vary_vit import Vary_VIT_B, Vary_VIT_B_Formula
|
||||
from .rec_pphgnetv2 import (
|
||||
PPHGNetV2_B4,
|
||||
PPHGNetV2_B4_Formula,
|
||||
PPHGNetV2_B6_Formula,
|
||||
)
|
||||
|
||||
support_dict = [
|
||||
"MobileNetV1Enhance",
|
||||
"MobileNetV3",
|
||||
"ResNet",
|
||||
"ResNetFPN",
|
||||
"MTB",
|
||||
"ResNet31",
|
||||
"ResNet45",
|
||||
"ResNet_ASTER",
|
||||
"MicroNet",
|
||||
"EfficientNetb3_PREN",
|
||||
"SVTRNet",
|
||||
"ViTSTR",
|
||||
"ResNet32",
|
||||
"ResNetRFL",
|
||||
"DenseNet",
|
||||
"ShallowCNN",
|
||||
"PPLCNetV3",
|
||||
"PPHGNet_small",
|
||||
"ViTParseQ",
|
||||
"ViT",
|
||||
"RepSVTR",
|
||||
"SVTRv2",
|
||||
"ResNetV2",
|
||||
"HybridTransformer",
|
||||
"DonutSwinModel",
|
||||
"Vary_VIT_B",
|
||||
"PPHGNetV2_B4",
|
||||
"PPHGNetV2_B4_Formula",
|
||||
"PPHGNetV2_B6_Formula",
|
||||
"Vary_VIT_B_Formula",
|
||||
]
|
||||
elif model_type == "e2e":
|
||||
from .e2e_resnet_vd_pg import ResNet
|
||||
|
||||
support_dict = ["ResNet"]
|
||||
elif model_type == "kie":
|
||||
from .kie_unet_sdmgr import Kie_backbone
|
||||
from .vqa_layoutlm import (
|
||||
LayoutLMForSer,
|
||||
LayoutLMv2ForSer,
|
||||
LayoutLMv2ForRe,
|
||||
LayoutXLMForSer,
|
||||
LayoutXLMForRe,
|
||||
)
|
||||
|
||||
support_dict = [
|
||||
"Kie_backbone",
|
||||
"LayoutLMForSer",
|
||||
"LayoutLMv2ForSer",
|
||||
"LayoutLMv2ForRe",
|
||||
"LayoutXLMForSer",
|
||||
"LayoutXLMForRe",
|
||||
]
|
||||
elif model_type == "table":
|
||||
from .table_resnet_vd import ResNet
|
||||
from .table_mobilenet_v3 import MobileNetV3
|
||||
from .rec_vary_vit import Vary_VIT_B
|
||||
|
||||
support_dict = ["ResNet", "MobileNetV3", "Vary_VIT_B"]
|
||||
else:
|
||||
raise NotImplementedError
|
||||
|
||||
module_name = config.pop("name")
|
||||
assert module_name in support_dict, Exception(
|
||||
"when model typs is {}, backbone only support {}".format(
|
||||
model_type, support_dict
|
||||
)
|
||||
)
|
||||
module_class = eval(module_name)(**config)
|
||||
return module_class
|
||||
289
ppocr/modeling/backbones/det_mobilenet_v3.py
Executable file
289
ppocr/modeling/backbones/det_mobilenet_v3.py
Executable file
@@ -0,0 +1,289 @@
|
||||
# copyright (c) 2020 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
import paddle.nn.functional as F
|
||||
from paddle import ParamAttr
|
||||
from ppocr.modeling.backbones.rec_hgnet import MeanPool2D
|
||||
|
||||
__all__ = ["MobileNetV3"]
|
||||
|
||||
|
||||
def make_divisible(v, divisor=8, min_value=None):
|
||||
if min_value is None:
|
||||
min_value = divisor
|
||||
new_v = max(min_value, int(v + divisor / 2) // divisor * divisor)
|
||||
if new_v < 0.9 * v:
|
||||
new_v += divisor
|
||||
return new_v
|
||||
|
||||
|
||||
class MobileNetV3(nn.Layer):
|
||||
def __init__(
|
||||
self, in_channels=3, model_name="large", scale=0.5, disable_se=False, **kwargs
|
||||
):
|
||||
"""
|
||||
the MobilenetV3 backbone network for detection module.
|
||||
Args:
|
||||
params(dict): the super parameters for build network
|
||||
"""
|
||||
super(MobileNetV3, self).__init__()
|
||||
|
||||
self.disable_se = disable_se
|
||||
|
||||
if model_name == "large":
|
||||
cfg = [
|
||||
# k, exp, c, se, nl, s,
|
||||
[3, 16, 16, False, "relu", 1],
|
||||
[3, 64, 24, False, "relu", 2],
|
||||
[3, 72, 24, False, "relu", 1],
|
||||
[5, 72, 40, True, "relu", 2],
|
||||
[5, 120, 40, True, "relu", 1],
|
||||
[5, 120, 40, True, "relu", 1],
|
||||
[3, 240, 80, False, "hardswish", 2],
|
||||
[3, 200, 80, False, "hardswish", 1],
|
||||
[3, 184, 80, False, "hardswish", 1],
|
||||
[3, 184, 80, False, "hardswish", 1],
|
||||
[3, 480, 112, True, "hardswish", 1],
|
||||
[3, 672, 112, True, "hardswish", 1],
|
||||
[5, 672, 160, True, "hardswish", 2],
|
||||
[5, 960, 160, True, "hardswish", 1],
|
||||
[5, 960, 160, True, "hardswish", 1],
|
||||
]
|
||||
cls_ch_squeeze = 960
|
||||
elif model_name == "small":
|
||||
cfg = [
|
||||
# k, exp, c, se, nl, s,
|
||||
[3, 16, 16, True, "relu", 2],
|
||||
[3, 72, 24, False, "relu", 2],
|
||||
[3, 88, 24, False, "relu", 1],
|
||||
[5, 96, 40, True, "hardswish", 2],
|
||||
[5, 240, 40, True, "hardswish", 1],
|
||||
[5, 240, 40, True, "hardswish", 1],
|
||||
[5, 120, 48, True, "hardswish", 1],
|
||||
[5, 144, 48, True, "hardswish", 1],
|
||||
[5, 288, 96, True, "hardswish", 2],
|
||||
[5, 576, 96, True, "hardswish", 1],
|
||||
[5, 576, 96, True, "hardswish", 1],
|
||||
]
|
||||
cls_ch_squeeze = 576
|
||||
else:
|
||||
raise NotImplementedError(
|
||||
"mode[" + model_name + "_model] is not implemented!"
|
||||
)
|
||||
|
||||
supported_scale = [0.35, 0.5, 0.75, 1.0, 1.25]
|
||||
assert (
|
||||
scale in supported_scale
|
||||
), "supported scale are {} but input scale is {}".format(supported_scale, scale)
|
||||
inplanes = 16
|
||||
# conv1
|
||||
self.conv = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=make_divisible(inplanes * scale),
|
||||
kernel_size=3,
|
||||
stride=2,
|
||||
padding=1,
|
||||
groups=1,
|
||||
if_act=True,
|
||||
act="hardswish",
|
||||
)
|
||||
|
||||
self.stages = []
|
||||
self.out_channels = []
|
||||
block_list = []
|
||||
i = 0
|
||||
inplanes = make_divisible(inplanes * scale)
|
||||
for k, exp, c, se, nl, s in cfg:
|
||||
se = se and not self.disable_se
|
||||
start_idx = 2 if model_name == "large" else 0
|
||||
if s == 2 and i > start_idx:
|
||||
self.out_channels.append(inplanes)
|
||||
self.stages.append(nn.Sequential(*block_list))
|
||||
block_list = []
|
||||
block_list.append(
|
||||
ResidualUnit(
|
||||
in_channels=inplanes,
|
||||
mid_channels=make_divisible(scale * exp),
|
||||
out_channels=make_divisible(scale * c),
|
||||
kernel_size=k,
|
||||
stride=s,
|
||||
use_se=se,
|
||||
act=nl,
|
||||
)
|
||||
)
|
||||
inplanes = make_divisible(scale * c)
|
||||
i += 1
|
||||
block_list.append(
|
||||
ConvBNLayer(
|
||||
in_channels=inplanes,
|
||||
out_channels=make_divisible(scale * cls_ch_squeeze),
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
padding=0,
|
||||
groups=1,
|
||||
if_act=True,
|
||||
act="hardswish",
|
||||
)
|
||||
)
|
||||
self.stages.append(nn.Sequential(*block_list))
|
||||
self.out_channels.append(make_divisible(scale * cls_ch_squeeze))
|
||||
for i, stage in enumerate(self.stages):
|
||||
self.add_sublayer(sublayer=stage, name="stage{}".format(i))
|
||||
|
||||
def forward(self, x):
|
||||
x = self.conv(x)
|
||||
out_list = []
|
||||
for stage in self.stages:
|
||||
x = stage(x)
|
||||
out_list.append(x)
|
||||
return out_list
|
||||
|
||||
|
||||
class ConvBNLayer(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
kernel_size,
|
||||
stride,
|
||||
padding,
|
||||
groups=1,
|
||||
if_act=True,
|
||||
act=None,
|
||||
):
|
||||
super(ConvBNLayer, self).__init__()
|
||||
self.if_act = if_act
|
||||
self.act = act
|
||||
self.conv = nn.Conv2D(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=kernel_size,
|
||||
stride=stride,
|
||||
padding=padding,
|
||||
groups=groups,
|
||||
bias_attr=False,
|
||||
)
|
||||
|
||||
self.bn = nn.BatchNorm(num_channels=out_channels, act=None)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.conv(x)
|
||||
x = self.bn(x)
|
||||
if self.if_act:
|
||||
if self.act == "relu":
|
||||
x = F.relu(x)
|
||||
elif self.act == "hardswish":
|
||||
x = F.hardswish(x)
|
||||
else:
|
||||
print(
|
||||
"The activation function({}) is selected incorrectly.".format(
|
||||
self.act
|
||||
)
|
||||
)
|
||||
exit()
|
||||
return x
|
||||
|
||||
|
||||
class ResidualUnit(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
mid_channels,
|
||||
out_channels,
|
||||
kernel_size,
|
||||
stride,
|
||||
use_se,
|
||||
act=None,
|
||||
):
|
||||
super(ResidualUnit, self).__init__()
|
||||
self.if_shortcut = stride == 1 and in_channels == out_channels
|
||||
self.if_se = use_se
|
||||
|
||||
self.expand_conv = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=mid_channels,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
padding=0,
|
||||
if_act=True,
|
||||
act=act,
|
||||
)
|
||||
self.bottleneck_conv = ConvBNLayer(
|
||||
in_channels=mid_channels,
|
||||
out_channels=mid_channels,
|
||||
kernel_size=kernel_size,
|
||||
stride=stride,
|
||||
padding=int((kernel_size - 1) // 2),
|
||||
groups=mid_channels,
|
||||
if_act=True,
|
||||
act=act,
|
||||
)
|
||||
if self.if_se:
|
||||
self.mid_se = SEModule(mid_channels)
|
||||
self.linear_conv = ConvBNLayer(
|
||||
in_channels=mid_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
padding=0,
|
||||
if_act=False,
|
||||
act=None,
|
||||
)
|
||||
|
||||
def forward(self, inputs):
|
||||
x = self.expand_conv(inputs)
|
||||
x = self.bottleneck_conv(x)
|
||||
if self.if_se:
|
||||
x = self.mid_se(x)
|
||||
x = self.linear_conv(x)
|
||||
if self.if_shortcut:
|
||||
x = paddle.add(inputs, x)
|
||||
return x
|
||||
|
||||
|
||||
class SEModule(nn.Layer):
|
||||
def __init__(self, in_channels, reduction=4):
|
||||
super(SEModule, self).__init__()
|
||||
if "npu" in paddle.device.get_device():
|
||||
self.avg_pool = MeanPool2D(1, 1)
|
||||
else:
|
||||
self.avg_pool = nn.AdaptiveAvgPool2D(1)
|
||||
self.conv1 = nn.Conv2D(
|
||||
in_channels=in_channels,
|
||||
out_channels=in_channels // reduction,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
padding=0,
|
||||
)
|
||||
self.conv2 = nn.Conv2D(
|
||||
in_channels=in_channels // reduction,
|
||||
out_channels=in_channels,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
padding=0,
|
||||
)
|
||||
|
||||
def forward(self, inputs):
|
||||
outputs = self.avg_pool(inputs)
|
||||
outputs = self.conv1(outputs)
|
||||
outputs = F.relu(outputs)
|
||||
outputs = self.conv2(outputs)
|
||||
outputs = F.hardsigmoid(outputs, slope=0.2, offset=0.5)
|
||||
return inputs * outputs
|
||||
274
ppocr/modeling/backbones/det_pp_lcnet.py
Normal file
274
ppocr/modeling/backbones/det_pp_lcnet.py
Normal file
@@ -0,0 +1,274 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import, division, print_function
|
||||
|
||||
import os
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
from paddle import ParamAttr
|
||||
from paddle.nn import AdaptiveAvgPool2D, BatchNorm, Conv2D, Dropout, Linear
|
||||
from paddle.regularizer import L2Decay
|
||||
from paddle.nn.initializer import KaimingNormal
|
||||
from paddle.utils.download import get_path_from_url
|
||||
|
||||
MODEL_URLS = {
|
||||
"PPLCNet_x0.25": "https://paddle-imagenet-models-name.bj.bcebos.com/dygraph/legendary_models/PPLCNet_x0_25_pretrained.pdparams",
|
||||
"PPLCNet_x0.35": "https://paddle-imagenet-models-name.bj.bcebos.com/dygraph/legendary_models/PPLCNet_x0_35_pretrained.pdparams",
|
||||
"PPLCNet_x0.5": "https://paddle-imagenet-models-name.bj.bcebos.com/dygraph/legendary_models/PPLCNet_x0_5_pretrained.pdparams",
|
||||
"PPLCNet_x0.75": "https://paddle-imagenet-models-name.bj.bcebos.com/dygraph/legendary_models/PPLCNet_x0_75_pretrained.pdparams",
|
||||
"PPLCNet_x1.0": "https://paddle-imagenet-models-name.bj.bcebos.com/dygraph/legendary_models/PPLCNet_x1_0_pretrained.pdparams",
|
||||
"PPLCNet_x1.5": "https://paddle-imagenet-models-name.bj.bcebos.com/dygraph/legendary_models/PPLCNet_x1_5_pretrained.pdparams",
|
||||
"PPLCNet_x2.0": "https://paddle-imagenet-models-name.bj.bcebos.com/dygraph/legendary_models/PPLCNet_x2_0_pretrained.pdparams",
|
||||
"PPLCNet_x2.5": "https://paddle-imagenet-models-name.bj.bcebos.com/dygraph/legendary_models/PPLCNet_x2_5_pretrained.pdparams",
|
||||
}
|
||||
|
||||
MODEL_STAGES_PATTERN = {
|
||||
"PPLCNet": ["blocks2", "blocks3", "blocks4", "blocks5", "blocks6"]
|
||||
}
|
||||
|
||||
__all__ = list(MODEL_URLS.keys())
|
||||
|
||||
# Each element(list) represents a depthwise block, which is composed of k, in_c, out_c, s, use_se.
|
||||
# k: kernel_size
|
||||
# in_c: input channel number in depthwise block
|
||||
# out_c: output channel number in depthwise block
|
||||
# s: stride in depthwise block
|
||||
# use_se: whether to use SE block
|
||||
|
||||
NET_CONFIG = {
|
||||
"blocks2":
|
||||
# k, in_c, out_c, s, use_se
|
||||
[[3, 16, 32, 1, False]],
|
||||
"blocks3": [[3, 32, 64, 2, False], [3, 64, 64, 1, False]],
|
||||
"blocks4": [[3, 64, 128, 2, False], [3, 128, 128, 1, False]],
|
||||
"blocks5": [
|
||||
[3, 128, 256, 2, False],
|
||||
[5, 256, 256, 1, False],
|
||||
[5, 256, 256, 1, False],
|
||||
[5, 256, 256, 1, False],
|
||||
[5, 256, 256, 1, False],
|
||||
[5, 256, 256, 1, False],
|
||||
],
|
||||
"blocks6": [[5, 256, 512, 2, True], [5, 512, 512, 1, True]],
|
||||
}
|
||||
|
||||
|
||||
def make_divisible(v, divisor=8, min_value=None):
|
||||
if min_value is None:
|
||||
min_value = divisor
|
||||
new_v = max(min_value, int(v + divisor / 2) // divisor * divisor)
|
||||
if new_v < 0.9 * v:
|
||||
new_v += divisor
|
||||
return new_v
|
||||
|
||||
|
||||
class ConvBNLayer(nn.Layer):
|
||||
def __init__(self, num_channels, filter_size, num_filters, stride, num_groups=1):
|
||||
super().__init__()
|
||||
|
||||
self.conv = Conv2D(
|
||||
in_channels=num_channels,
|
||||
out_channels=num_filters,
|
||||
kernel_size=filter_size,
|
||||
stride=stride,
|
||||
padding=(filter_size - 1) // 2,
|
||||
groups=num_groups,
|
||||
weight_attr=ParamAttr(initializer=KaimingNormal()),
|
||||
bias_attr=False,
|
||||
)
|
||||
|
||||
self.bn = BatchNorm(
|
||||
num_filters,
|
||||
param_attr=ParamAttr(regularizer=L2Decay(0.0)),
|
||||
bias_attr=ParamAttr(regularizer=L2Decay(0.0)),
|
||||
)
|
||||
self.hardswish = nn.Hardswish()
|
||||
|
||||
def forward(self, x):
|
||||
x = self.conv(x)
|
||||
x = self.bn(x)
|
||||
x = self.hardswish(x)
|
||||
return x
|
||||
|
||||
|
||||
class DepthwiseSeparable(nn.Layer):
|
||||
def __init__(self, num_channels, num_filters, stride, dw_size=3, use_se=False):
|
||||
super().__init__()
|
||||
self.use_se = use_se
|
||||
self.dw_conv = ConvBNLayer(
|
||||
num_channels=num_channels,
|
||||
num_filters=num_channels,
|
||||
filter_size=dw_size,
|
||||
stride=stride,
|
||||
num_groups=num_channels,
|
||||
)
|
||||
if use_se:
|
||||
self.se = SEModule(num_channels)
|
||||
self.pw_conv = ConvBNLayer(
|
||||
num_channels=num_channels, filter_size=1, num_filters=num_filters, stride=1
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.dw_conv(x)
|
||||
if self.use_se:
|
||||
x = self.se(x)
|
||||
x = self.pw_conv(x)
|
||||
return x
|
||||
|
||||
|
||||
class SEModule(nn.Layer):
|
||||
def __init__(self, channel, reduction=4):
|
||||
super().__init__()
|
||||
self.avg_pool = AdaptiveAvgPool2D(1)
|
||||
self.conv1 = Conv2D(
|
||||
in_channels=channel,
|
||||
out_channels=channel // reduction,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
padding=0,
|
||||
)
|
||||
self.relu = nn.ReLU()
|
||||
self.conv2 = Conv2D(
|
||||
in_channels=channel // reduction,
|
||||
out_channels=channel,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
padding=0,
|
||||
)
|
||||
self.hardsigmoid = nn.Hardsigmoid()
|
||||
|
||||
def forward(self, x):
|
||||
identity = x
|
||||
x = self.avg_pool(x)
|
||||
x = self.conv1(x)
|
||||
x = self.relu(x)
|
||||
x = self.conv2(x)
|
||||
x = self.hardsigmoid(x)
|
||||
x = paddle.multiply(x=identity, y=x)
|
||||
return x
|
||||
|
||||
|
||||
class PPLCNet(nn.Layer):
|
||||
def __init__(self, in_channels=3, scale=1.0, pretrained=False, use_ssld=False):
|
||||
super().__init__()
|
||||
self.out_channels = [
|
||||
int(NET_CONFIG["blocks3"][-1][2] * scale),
|
||||
int(NET_CONFIG["blocks4"][-1][2] * scale),
|
||||
int(NET_CONFIG["blocks5"][-1][2] * scale),
|
||||
int(NET_CONFIG["blocks6"][-1][2] * scale),
|
||||
]
|
||||
self.scale = scale
|
||||
|
||||
self.conv1 = ConvBNLayer(
|
||||
num_channels=in_channels,
|
||||
filter_size=3,
|
||||
num_filters=make_divisible(16 * scale),
|
||||
stride=2,
|
||||
)
|
||||
|
||||
self.blocks2 = nn.Sequential(
|
||||
*[
|
||||
DepthwiseSeparable(
|
||||
num_channels=make_divisible(in_c * scale),
|
||||
num_filters=make_divisible(out_c * scale),
|
||||
dw_size=k,
|
||||
stride=s,
|
||||
use_se=se,
|
||||
)
|
||||
for i, (k, in_c, out_c, s, se) in enumerate(NET_CONFIG["blocks2"])
|
||||
]
|
||||
)
|
||||
|
||||
self.blocks3 = nn.Sequential(
|
||||
*[
|
||||
DepthwiseSeparable(
|
||||
num_channels=make_divisible(in_c * scale),
|
||||
num_filters=make_divisible(out_c * scale),
|
||||
dw_size=k,
|
||||
stride=s,
|
||||
use_se=se,
|
||||
)
|
||||
for i, (k, in_c, out_c, s, se) in enumerate(NET_CONFIG["blocks3"])
|
||||
]
|
||||
)
|
||||
|
||||
self.blocks4 = nn.Sequential(
|
||||
*[
|
||||
DepthwiseSeparable(
|
||||
num_channels=make_divisible(in_c * scale),
|
||||
num_filters=make_divisible(out_c * scale),
|
||||
dw_size=k,
|
||||
stride=s,
|
||||
use_se=se,
|
||||
)
|
||||
for i, (k, in_c, out_c, s, se) in enumerate(NET_CONFIG["blocks4"])
|
||||
]
|
||||
)
|
||||
|
||||
self.blocks5 = nn.Sequential(
|
||||
*[
|
||||
DepthwiseSeparable(
|
||||
num_channels=make_divisible(in_c * scale),
|
||||
num_filters=make_divisible(out_c * scale),
|
||||
dw_size=k,
|
||||
stride=s,
|
||||
use_se=se,
|
||||
)
|
||||
for i, (k, in_c, out_c, s, se) in enumerate(NET_CONFIG["blocks5"])
|
||||
]
|
||||
)
|
||||
|
||||
self.blocks6 = nn.Sequential(
|
||||
*[
|
||||
DepthwiseSeparable(
|
||||
num_channels=make_divisible(in_c * scale),
|
||||
num_filters=make_divisible(out_c * scale),
|
||||
dw_size=k,
|
||||
stride=s,
|
||||
use_se=se,
|
||||
)
|
||||
for i, (k, in_c, out_c, s, se) in enumerate(NET_CONFIG["blocks6"])
|
||||
]
|
||||
)
|
||||
|
||||
if pretrained:
|
||||
self._load_pretrained(
|
||||
MODEL_URLS["PPLCNet_x{}".format(scale)], use_ssld=use_ssld
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
outs = []
|
||||
x = self.conv1(x)
|
||||
x = self.blocks2(x)
|
||||
x = self.blocks3(x)
|
||||
outs.append(x)
|
||||
x = self.blocks4(x)
|
||||
outs.append(x)
|
||||
x = self.blocks5(x)
|
||||
outs.append(x)
|
||||
x = self.blocks6(x)
|
||||
outs.append(x)
|
||||
return outs
|
||||
|
||||
def _load_pretrained(self, pretrained_url, use_ssld=False):
|
||||
if use_ssld:
|
||||
pretrained_url = pretrained_url.replace("_pretrained", "_ssld_pretrained")
|
||||
print(pretrained_url)
|
||||
local_weight_path = get_path_from_url(
|
||||
pretrained_url, os.path.expanduser("~/.paddleclas/weights")
|
||||
)
|
||||
param_state_dict = paddle.load(local_weight_path)
|
||||
self.set_dict(param_state_dict)
|
||||
return
|
||||
358
ppocr/modeling/backbones/det_pp_lcnet_v2.py
Normal file
358
ppocr/modeling/backbones/det_pp_lcnet_v2.py
Normal file
@@ -0,0 +1,358 @@
|
||||
# copyright (c) 2024 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import, division, print_function
|
||||
import os
|
||||
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
from paddle import ParamAttr
|
||||
from paddle.nn import AdaptiveAvgPool2D, BatchNorm2D, Conv2D, Dropout, Linear
|
||||
from paddle.regularizer import L2Decay
|
||||
from paddle.nn.initializer import KaimingNormal
|
||||
from paddle.utils.download import get_path_from_url
|
||||
|
||||
MODEL_URLS = {
|
||||
"PPLCNetV2_small": "https://paddle-imagenet-models-name.bj.bcebos.com/dygraph/legendary_models/PPLCNetV2_small_ssld_pretrained.pdparams",
|
||||
"PPLCNetV2_base": "https://paddle-imagenet-models-name.bj.bcebos.com/dygraph/legendary_models/PPLCNetV2_base_ssld_pretrained.pdparams",
|
||||
"PPLCNetV2_large": "https://paddle-imagenet-models-name.bj.bcebos.com/dygraph/legendary_models/PPLCNetV2_large_ssld_pretrained.pdparams",
|
||||
}
|
||||
|
||||
__all__ = list(MODEL_URLS.keys())
|
||||
|
||||
NET_CONFIG = {
|
||||
# in_channels, kernel_size, split_pw, use_rep, use_se, use_shortcut
|
||||
"stage1": [64, 3, False, False, False, False],
|
||||
"stage2": [128, 3, False, False, False, False],
|
||||
"stage3": [256, 5, True, True, True, False],
|
||||
"stage4": [512, 5, False, True, False, True],
|
||||
}
|
||||
|
||||
|
||||
def make_divisible(v, divisor=8, min_value=None):
|
||||
if min_value is None:
|
||||
min_value = divisor
|
||||
new_v = max(min_value, int(v + divisor / 2) // divisor * divisor)
|
||||
if new_v < 0.9 * v:
|
||||
new_v += divisor
|
||||
return new_v
|
||||
|
||||
|
||||
class ConvBNLayer(nn.Layer):
|
||||
def __init__(
|
||||
self, in_channels, out_channels, kernel_size, stride, groups=1, use_act=True
|
||||
):
|
||||
super().__init__()
|
||||
self.use_act = use_act
|
||||
self.conv = Conv2D(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=kernel_size,
|
||||
stride=stride,
|
||||
padding=(kernel_size - 1) // 2,
|
||||
groups=groups,
|
||||
weight_attr=ParamAttr(initializer=KaimingNormal()),
|
||||
bias_attr=False,
|
||||
)
|
||||
|
||||
self.bn = BatchNorm2D(
|
||||
out_channels,
|
||||
weight_attr=ParamAttr(regularizer=L2Decay(0.0)),
|
||||
bias_attr=ParamAttr(regularizer=L2Decay(0.0)),
|
||||
)
|
||||
if self.use_act:
|
||||
self.act = nn.ReLU()
|
||||
|
||||
def forward(self, x):
|
||||
x = self.conv(x)
|
||||
x = self.bn(x)
|
||||
if self.use_act:
|
||||
x = self.act(x)
|
||||
return x
|
||||
|
||||
|
||||
class SEModule(nn.Layer):
|
||||
def __init__(self, channel, reduction=4):
|
||||
super().__init__()
|
||||
self.avg_pool = AdaptiveAvgPool2D(1)
|
||||
self.conv1 = Conv2D(
|
||||
in_channels=channel,
|
||||
out_channels=channel // reduction,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
padding=0,
|
||||
)
|
||||
self.relu = nn.ReLU()
|
||||
self.conv2 = Conv2D(
|
||||
in_channels=channel // reduction,
|
||||
out_channels=channel,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
padding=0,
|
||||
)
|
||||
self.hardsigmoid = nn.Sigmoid()
|
||||
|
||||
def forward(self, x):
|
||||
identity = x
|
||||
x = self.avg_pool(x)
|
||||
x = self.conv1(x)
|
||||
x = self.relu(x)
|
||||
x = self.conv2(x)
|
||||
x = self.hardsigmoid(x)
|
||||
x = paddle.multiply(x=identity, y=x)
|
||||
return x
|
||||
|
||||
|
||||
class RepDepthwiseSeparable(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
stride,
|
||||
dw_size=3,
|
||||
split_pw=False,
|
||||
use_rep=False,
|
||||
use_se=False,
|
||||
use_shortcut=False,
|
||||
):
|
||||
super().__init__()
|
||||
self.in_channels = in_channels
|
||||
self.out_channels = out_channels
|
||||
self.is_repped = False
|
||||
|
||||
self.dw_size = dw_size
|
||||
self.split_pw = split_pw
|
||||
self.use_rep = use_rep
|
||||
self.use_se = use_se
|
||||
self.use_shortcut = (
|
||||
True
|
||||
if use_shortcut and stride == 1 and in_channels == out_channels
|
||||
else False
|
||||
)
|
||||
|
||||
if self.use_rep:
|
||||
self.dw_conv_list = nn.LayerList()
|
||||
for kernel_size in range(self.dw_size, 0, -2):
|
||||
if kernel_size == 1 and stride != 1:
|
||||
continue
|
||||
dw_conv = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=in_channels,
|
||||
kernel_size=kernel_size,
|
||||
stride=stride,
|
||||
groups=in_channels,
|
||||
use_act=False,
|
||||
)
|
||||
self.dw_conv_list.append(dw_conv)
|
||||
self.dw_conv = nn.Conv2D(
|
||||
in_channels=in_channels,
|
||||
out_channels=in_channels,
|
||||
kernel_size=dw_size,
|
||||
stride=stride,
|
||||
padding=(dw_size - 1) // 2,
|
||||
groups=in_channels,
|
||||
)
|
||||
else:
|
||||
self.dw_conv = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=in_channels,
|
||||
kernel_size=dw_size,
|
||||
stride=stride,
|
||||
groups=in_channels,
|
||||
)
|
||||
|
||||
self.act = nn.ReLU()
|
||||
|
||||
if use_se:
|
||||
self.se = SEModule(in_channels)
|
||||
|
||||
if self.split_pw:
|
||||
pw_ratio = 0.5
|
||||
self.pw_conv_1 = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
kernel_size=1,
|
||||
out_channels=int(out_channels * pw_ratio),
|
||||
stride=1,
|
||||
)
|
||||
self.pw_conv_2 = ConvBNLayer(
|
||||
in_channels=int(out_channels * pw_ratio),
|
||||
kernel_size=1,
|
||||
out_channels=out_channels,
|
||||
stride=1,
|
||||
)
|
||||
else:
|
||||
self.pw_conv = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
kernel_size=1,
|
||||
out_channels=out_channels,
|
||||
stride=1,
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
if self.use_rep:
|
||||
input_x = x
|
||||
if self.is_repped:
|
||||
x = self.act(self.dw_conv(x))
|
||||
else:
|
||||
y = self.dw_conv_list[0](x)
|
||||
for dw_conv in self.dw_conv_list[1:]:
|
||||
y += dw_conv(x)
|
||||
x = self.act(y)
|
||||
else:
|
||||
x = self.dw_conv(x)
|
||||
|
||||
if self.use_se:
|
||||
x = self.se(x)
|
||||
if self.split_pw:
|
||||
x = self.pw_conv_1(x)
|
||||
x = self.pw_conv_2(x)
|
||||
else:
|
||||
x = self.pw_conv(x)
|
||||
if self.use_shortcut:
|
||||
x = x + input_x
|
||||
return x
|
||||
|
||||
def re_parameterize(self):
|
||||
if self.use_rep:
|
||||
self.is_repped = True
|
||||
kernel, bias = self._get_equivalent_kernel_bias()
|
||||
self.dw_conv.weight.set_value(kernel)
|
||||
self.dw_conv.bias.set_value(bias)
|
||||
|
||||
def _get_equivalent_kernel_bias(self):
|
||||
kernel_sum = 0
|
||||
bias_sum = 0
|
||||
for dw_conv in self.dw_conv_list:
|
||||
kernel, bias = self._fuse_bn_tensor(dw_conv)
|
||||
kernel = self._pad_tensor(kernel, to_size=self.dw_size)
|
||||
kernel_sum += kernel
|
||||
bias_sum += bias
|
||||
return kernel_sum, bias_sum
|
||||
|
||||
def _fuse_bn_tensor(self, branch):
|
||||
kernel = branch.conv.weight
|
||||
running_mean = branch.bn._mean
|
||||
running_var = branch.bn._variance
|
||||
gamma = branch.bn.weight
|
||||
beta = branch.bn.bias
|
||||
eps = branch.bn._epsilon
|
||||
std = (running_var + eps).sqrt()
|
||||
t = (gamma / std).reshape((-1, 1, 1, 1))
|
||||
return kernel * t, beta - running_mean * gamma / std
|
||||
|
||||
def _pad_tensor(self, tensor, to_size):
|
||||
from_size = tensor.shape[-1]
|
||||
if from_size == to_size:
|
||||
return tensor
|
||||
pad = (to_size - from_size) // 2
|
||||
return F.pad(tensor, [pad, pad, pad, pad])
|
||||
|
||||
|
||||
class PPLCNetV2(nn.Layer):
|
||||
def __init__(self, scale, depths, out_indx=[1, 2, 3, 4], **kwargs):
|
||||
super().__init__(**kwargs)
|
||||
self.scale = scale
|
||||
self.out_channels = [
|
||||
# int(NET_CONFIG["blocks3"][-1][2] * scale),
|
||||
int(NET_CONFIG["stage1"][0] * scale * 2),
|
||||
int(NET_CONFIG["stage2"][0] * scale * 2),
|
||||
int(NET_CONFIG["stage3"][0] * scale * 2),
|
||||
int(NET_CONFIG["stage4"][0] * scale * 2),
|
||||
]
|
||||
self.stem = nn.Sequential(
|
||||
*[
|
||||
ConvBNLayer(
|
||||
in_channels=3,
|
||||
kernel_size=3,
|
||||
out_channels=make_divisible(32 * scale),
|
||||
stride=2,
|
||||
),
|
||||
RepDepthwiseSeparable(
|
||||
in_channels=make_divisible(32 * scale),
|
||||
out_channels=make_divisible(64 * scale),
|
||||
stride=1,
|
||||
dw_size=3,
|
||||
),
|
||||
]
|
||||
)
|
||||
self.out_indx = out_indx
|
||||
# stages
|
||||
self.stages = nn.LayerList()
|
||||
for depth_idx, k in enumerate(NET_CONFIG):
|
||||
(
|
||||
in_channels,
|
||||
kernel_size,
|
||||
split_pw,
|
||||
use_rep,
|
||||
use_se,
|
||||
use_shortcut,
|
||||
) = NET_CONFIG[k]
|
||||
self.stages.append(
|
||||
nn.Sequential(
|
||||
*[
|
||||
RepDepthwiseSeparable(
|
||||
in_channels=make_divisible(
|
||||
(in_channels if i == 0 else in_channels * 2) * scale
|
||||
),
|
||||
out_channels=make_divisible(in_channels * 2 * scale),
|
||||
stride=2 if i == 0 else 1,
|
||||
dw_size=kernel_size,
|
||||
split_pw=split_pw,
|
||||
use_rep=use_rep,
|
||||
use_se=use_se,
|
||||
use_shortcut=use_shortcut,
|
||||
)
|
||||
for i in range(depths[depth_idx])
|
||||
]
|
||||
)
|
||||
)
|
||||
|
||||
# if pretrained:
|
||||
self._load_pretrained(MODEL_URLS["PPLCNetV2_base"], use_ssld=True)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.stem(x)
|
||||
i = 1
|
||||
outs = []
|
||||
for stage in self.stages:
|
||||
x = stage(x)
|
||||
if i in self.out_indx:
|
||||
outs.append(x)
|
||||
i += 1
|
||||
return outs
|
||||
|
||||
def _load_pretrained(self, pretrained_url, use_ssld=False):
|
||||
print(pretrained_url)
|
||||
local_weight_path = get_path_from_url(
|
||||
pretrained_url, os.path.expanduser("~/.paddleclas/weights")
|
||||
)
|
||||
param_state_dict = paddle.load(local_weight_path)
|
||||
self.set_dict(param_state_dict)
|
||||
print("load pretrain ssd success!")
|
||||
return
|
||||
|
||||
|
||||
def PPLCNetV2_base(in_channels=3, **kwargs):
|
||||
"""
|
||||
PPLCNetV2_base
|
||||
Args:
|
||||
pretrained: bool=False or str. If `True` load pretrained parameters, `False` otherwise.
|
||||
If str, means the path of the pretrained model.
|
||||
use_ssld: bool=False. Whether using distillation pretrained model when pretrained=True.
|
||||
Returns:
|
||||
model: nn.Layer. Specific `PPLCNetV2_base` model depends on args.
|
||||
"""
|
||||
model = PPLCNetV2(scale=1.0, depths=[2, 2, 6, 2], **kwargs)
|
||||
return model
|
||||
235
ppocr/modeling/backbones/det_resnet.py
Normal file
235
ppocr/modeling/backbones/det_resnet.py
Normal file
@@ -0,0 +1,235 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import numpy as np
|
||||
import paddle
|
||||
from paddle import ParamAttr
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
from paddle.nn import Conv2D, BatchNorm, Linear, Dropout
|
||||
from paddle.nn import AdaptiveAvgPool2D, MaxPool2D, AvgPool2D
|
||||
from paddle.nn.initializer import Uniform
|
||||
|
||||
import math
|
||||
|
||||
from paddle.vision.ops import DeformConv2D
|
||||
from paddle.regularizer import L2Decay
|
||||
from paddle.nn.initializer import Normal, Constant, XavierUniform
|
||||
from .det_resnet_vd import DeformableConvV2, ConvBNLayer
|
||||
|
||||
|
||||
class BottleneckBlock(nn.Layer):
|
||||
def __init__(self, num_channels, num_filters, stride, shortcut=True, is_dcn=False):
|
||||
super(BottleneckBlock, self).__init__()
|
||||
|
||||
self.conv0 = ConvBNLayer(
|
||||
in_channels=num_channels,
|
||||
out_channels=num_filters,
|
||||
kernel_size=1,
|
||||
act="relu",
|
||||
)
|
||||
self.conv1 = ConvBNLayer(
|
||||
in_channels=num_filters,
|
||||
out_channels=num_filters,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
act="relu",
|
||||
is_dcn=is_dcn,
|
||||
dcn_groups=1,
|
||||
)
|
||||
self.conv2 = ConvBNLayer(
|
||||
in_channels=num_filters,
|
||||
out_channels=num_filters * 4,
|
||||
kernel_size=1,
|
||||
act=None,
|
||||
)
|
||||
|
||||
if not shortcut:
|
||||
self.short = ConvBNLayer(
|
||||
in_channels=num_channels,
|
||||
out_channels=num_filters * 4,
|
||||
kernel_size=1,
|
||||
stride=stride,
|
||||
)
|
||||
|
||||
self.shortcut = shortcut
|
||||
|
||||
self._num_channels_out = num_filters * 4
|
||||
|
||||
def forward(self, inputs):
|
||||
y = self.conv0(inputs)
|
||||
conv1 = self.conv1(y)
|
||||
conv2 = self.conv2(conv1)
|
||||
|
||||
if self.shortcut:
|
||||
short = inputs
|
||||
else:
|
||||
short = self.short(inputs)
|
||||
|
||||
y = paddle.add(x=short, y=conv2)
|
||||
y = F.relu(y)
|
||||
return y
|
||||
|
||||
|
||||
class BasicBlock(nn.Layer):
|
||||
def __init__(self, num_channels, num_filters, stride, shortcut=True, name=None):
|
||||
super(BasicBlock, self).__init__()
|
||||
self.stride = stride
|
||||
self.conv0 = ConvBNLayer(
|
||||
in_channels=num_channels,
|
||||
out_channels=num_filters,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
act="relu",
|
||||
)
|
||||
self.conv1 = ConvBNLayer(
|
||||
in_channels=num_filters, out_channels=num_filters, kernel_size=3, act=None
|
||||
)
|
||||
|
||||
if not shortcut:
|
||||
self.short = ConvBNLayer(
|
||||
in_channels=num_channels,
|
||||
out_channels=num_filters,
|
||||
kernel_size=1,
|
||||
stride=stride,
|
||||
)
|
||||
|
||||
self.shortcut = shortcut
|
||||
|
||||
def forward(self, inputs):
|
||||
y = self.conv0(inputs)
|
||||
conv1 = self.conv1(y)
|
||||
|
||||
if self.shortcut:
|
||||
short = inputs
|
||||
else:
|
||||
short = self.short(inputs)
|
||||
y = paddle.add(x=short, y=conv1)
|
||||
y = F.relu(y)
|
||||
return y
|
||||
|
||||
|
||||
class ResNet(nn.Layer):
|
||||
def __init__(self, in_channels=3, layers=50, out_indices=None, dcn_stage=None):
|
||||
super(ResNet, self).__init__()
|
||||
|
||||
self.layers = layers
|
||||
self.input_image_channel = in_channels
|
||||
|
||||
supported_layers = [18, 34, 50, 101, 152]
|
||||
assert (
|
||||
layers in supported_layers
|
||||
), "supported layers are {} but input layer is {}".format(
|
||||
supported_layers, layers
|
||||
)
|
||||
|
||||
if layers == 18:
|
||||
depth = [2, 2, 2, 2]
|
||||
elif layers == 34 or layers == 50:
|
||||
depth = [3, 4, 6, 3]
|
||||
elif layers == 101:
|
||||
depth = [3, 4, 23, 3]
|
||||
elif layers == 152:
|
||||
depth = [3, 8, 36, 3]
|
||||
num_channels = [64, 256, 512, 1024] if layers >= 50 else [64, 64, 128, 256]
|
||||
num_filters = [64, 128, 256, 512]
|
||||
|
||||
self.dcn_stage = (
|
||||
dcn_stage if dcn_stage is not None else [False, False, False, False]
|
||||
)
|
||||
self.out_indices = out_indices if out_indices is not None else [0, 1, 2, 3]
|
||||
|
||||
self.conv = ConvBNLayer(
|
||||
in_channels=self.input_image_channel,
|
||||
out_channels=64,
|
||||
kernel_size=7,
|
||||
stride=2,
|
||||
act="relu",
|
||||
)
|
||||
self.pool2d_max = MaxPool2D(
|
||||
kernel_size=3,
|
||||
stride=2,
|
||||
padding=1,
|
||||
)
|
||||
|
||||
self.stages = []
|
||||
self.out_channels = []
|
||||
if layers >= 50:
|
||||
for block in range(len(depth)):
|
||||
shortcut = False
|
||||
block_list = []
|
||||
is_dcn = self.dcn_stage[block]
|
||||
for i in range(depth[block]):
|
||||
if layers in [101, 152] and block == 2:
|
||||
if i == 0:
|
||||
conv_name = "res" + str(block + 2) + "a"
|
||||
else:
|
||||
conv_name = "res" + str(block + 2) + "b" + str(i)
|
||||
else:
|
||||
conv_name = "res" + str(block + 2) + chr(97 + i)
|
||||
bottleneck_block = self.add_sublayer(
|
||||
conv_name,
|
||||
BottleneckBlock(
|
||||
num_channels=(
|
||||
num_channels[block]
|
||||
if i == 0
|
||||
else num_filters[block] * 4
|
||||
),
|
||||
num_filters=num_filters[block],
|
||||
stride=2 if i == 0 and block != 0 else 1,
|
||||
shortcut=shortcut,
|
||||
is_dcn=is_dcn,
|
||||
),
|
||||
)
|
||||
block_list.append(bottleneck_block)
|
||||
shortcut = True
|
||||
if block in self.out_indices:
|
||||
self.out_channels.append(num_filters[block] * 4)
|
||||
self.stages.append(nn.Sequential(*block_list))
|
||||
else:
|
||||
for block in range(len(depth)):
|
||||
shortcut = False
|
||||
block_list = []
|
||||
for i in range(depth[block]):
|
||||
conv_name = "res" + str(block + 2) + chr(97 + i)
|
||||
basic_block = self.add_sublayer(
|
||||
conv_name,
|
||||
BasicBlock(
|
||||
num_channels=(
|
||||
num_channels[block] if i == 0 else num_filters[block]
|
||||
),
|
||||
num_filters=num_filters[block],
|
||||
stride=2 if i == 0 and block != 0 else 1,
|
||||
shortcut=shortcut,
|
||||
),
|
||||
)
|
||||
block_list.append(basic_block)
|
||||
shortcut = True
|
||||
if block in self.out_indices:
|
||||
self.out_channels.append(num_filters[block])
|
||||
self.stages.append(nn.Sequential(*block_list))
|
||||
|
||||
def forward(self, inputs):
|
||||
y = self.conv(inputs)
|
||||
y = self.pool2d_max(y)
|
||||
out = []
|
||||
for i, block in enumerate(self.stages):
|
||||
y = block(y)
|
||||
if i in self.out_indices:
|
||||
out.append(y)
|
||||
return out
|
||||
369
ppocr/modeling/backbones/det_resnet_vd.py
Normal file
369
ppocr/modeling/backbones/det_resnet_vd.py
Normal file
@@ -0,0 +1,369 @@
|
||||
# copyright (c) 2020 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import ParamAttr
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
|
||||
from paddle.vision.ops import DeformConv2D
|
||||
from paddle.regularizer import L2Decay
|
||||
from paddle.nn.initializer import Normal, Constant, XavierUniform
|
||||
|
||||
__all__ = ["ResNet_vd", "ConvBNLayer", "DeformableConvV2"]
|
||||
|
||||
|
||||
class DeformableConvV2(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
kernel_size,
|
||||
stride=1,
|
||||
padding=0,
|
||||
dilation=1,
|
||||
groups=1,
|
||||
weight_attr=None,
|
||||
bias_attr=None,
|
||||
lr_scale=1,
|
||||
regularizer=None,
|
||||
skip_quant=False,
|
||||
dcn_bias_regularizer=L2Decay(0.0),
|
||||
dcn_bias_lr_scale=2.0,
|
||||
):
|
||||
super(DeformableConvV2, self).__init__()
|
||||
self.offset_channel = 2 * kernel_size**2 * groups
|
||||
self.mask_channel = kernel_size**2 * groups
|
||||
|
||||
if bias_attr:
|
||||
# in FCOS-DCN head, specifically need learning_rate and regularizer
|
||||
dcn_bias_attr = ParamAttr(
|
||||
initializer=Constant(value=0),
|
||||
regularizer=dcn_bias_regularizer,
|
||||
learning_rate=dcn_bias_lr_scale,
|
||||
)
|
||||
else:
|
||||
# in ResNet backbone, do not need bias
|
||||
dcn_bias_attr = False
|
||||
self.conv_dcn = DeformConv2D(
|
||||
in_channels,
|
||||
out_channels,
|
||||
kernel_size,
|
||||
stride=stride,
|
||||
padding=(kernel_size - 1) // 2 * dilation,
|
||||
dilation=dilation,
|
||||
deformable_groups=groups,
|
||||
weight_attr=weight_attr,
|
||||
bias_attr=dcn_bias_attr,
|
||||
)
|
||||
|
||||
if lr_scale == 1 and regularizer is None:
|
||||
offset_bias_attr = ParamAttr(initializer=Constant(0.0))
|
||||
else:
|
||||
offset_bias_attr = ParamAttr(
|
||||
initializer=Constant(0.0),
|
||||
learning_rate=lr_scale,
|
||||
regularizer=regularizer,
|
||||
)
|
||||
self.conv_offset = nn.Conv2D(
|
||||
in_channels,
|
||||
groups * 3 * kernel_size**2,
|
||||
kernel_size,
|
||||
stride=stride,
|
||||
padding=(kernel_size - 1) // 2,
|
||||
weight_attr=ParamAttr(initializer=Constant(0.0)),
|
||||
bias_attr=offset_bias_attr,
|
||||
)
|
||||
if skip_quant:
|
||||
self.conv_offset.skip_quant = True
|
||||
|
||||
def forward(self, x):
|
||||
offset_mask = self.conv_offset(x)
|
||||
offset, mask = paddle.split(
|
||||
offset_mask,
|
||||
num_or_sections=[self.offset_channel, self.mask_channel],
|
||||
axis=1,
|
||||
)
|
||||
mask = F.sigmoid(mask)
|
||||
y = self.conv_dcn(x, offset, mask=mask)
|
||||
return y
|
||||
|
||||
|
||||
class ConvBNLayer(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
kernel_size,
|
||||
stride=1,
|
||||
groups=1,
|
||||
dcn_groups=1,
|
||||
is_vd_mode=False,
|
||||
act=None,
|
||||
is_dcn=False,
|
||||
):
|
||||
super(ConvBNLayer, self).__init__()
|
||||
|
||||
self.is_vd_mode = is_vd_mode
|
||||
self._pool2d_avg = nn.AvgPool2D(
|
||||
kernel_size=2, stride=2, padding=0, ceil_mode=True
|
||||
)
|
||||
if not is_dcn:
|
||||
self._conv = nn.Conv2D(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=kernel_size,
|
||||
stride=stride,
|
||||
padding=(kernel_size - 1) // 2,
|
||||
groups=groups,
|
||||
bias_attr=False,
|
||||
)
|
||||
else:
|
||||
self._conv = DeformableConvV2(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=kernel_size,
|
||||
stride=stride,
|
||||
padding=(kernel_size - 1) // 2,
|
||||
groups=dcn_groups, # groups,
|
||||
bias_attr=False,
|
||||
)
|
||||
self._batch_norm = nn.BatchNorm(out_channels, act=act)
|
||||
|
||||
def forward(self, inputs):
|
||||
if self.is_vd_mode:
|
||||
inputs = self._pool2d_avg(inputs)
|
||||
y = self._conv(inputs)
|
||||
y = self._batch_norm(y)
|
||||
return y
|
||||
|
||||
|
||||
class BottleneckBlock(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
stride,
|
||||
shortcut=True,
|
||||
if_first=False,
|
||||
is_dcn=False,
|
||||
):
|
||||
super(BottleneckBlock, self).__init__()
|
||||
|
||||
self.conv0 = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=1,
|
||||
act="relu",
|
||||
)
|
||||
self.conv1 = ConvBNLayer(
|
||||
in_channels=out_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
act="relu",
|
||||
is_dcn=is_dcn,
|
||||
dcn_groups=2,
|
||||
)
|
||||
self.conv2 = ConvBNLayer(
|
||||
in_channels=out_channels,
|
||||
out_channels=out_channels * 4,
|
||||
kernel_size=1,
|
||||
act=None,
|
||||
)
|
||||
|
||||
if not shortcut:
|
||||
self.short = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels * 4,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
is_vd_mode=False if if_first else True,
|
||||
)
|
||||
|
||||
self.shortcut = shortcut
|
||||
|
||||
def forward(self, inputs):
|
||||
y = self.conv0(inputs)
|
||||
conv1 = self.conv1(y)
|
||||
conv2 = self.conv2(conv1)
|
||||
|
||||
if self.shortcut:
|
||||
short = inputs
|
||||
else:
|
||||
short = self.short(inputs)
|
||||
y = paddle.add(x=short, y=conv2)
|
||||
y = F.relu(y)
|
||||
return y
|
||||
|
||||
|
||||
class BasicBlock(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
stride,
|
||||
shortcut=True,
|
||||
if_first=False,
|
||||
):
|
||||
super(BasicBlock, self).__init__()
|
||||
self.stride = stride
|
||||
self.conv0 = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
act="relu",
|
||||
)
|
||||
self.conv1 = ConvBNLayer(
|
||||
in_channels=out_channels, out_channels=out_channels, kernel_size=3, act=None
|
||||
)
|
||||
|
||||
if not shortcut:
|
||||
self.short = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
is_vd_mode=False if if_first else True,
|
||||
)
|
||||
|
||||
self.shortcut = shortcut
|
||||
|
||||
def forward(self, inputs):
|
||||
y = self.conv0(inputs)
|
||||
conv1 = self.conv1(y)
|
||||
|
||||
if self.shortcut:
|
||||
short = inputs
|
||||
else:
|
||||
short = self.short(inputs)
|
||||
y = paddle.add(x=short, y=conv1)
|
||||
y = F.relu(y)
|
||||
return y
|
||||
|
||||
|
||||
class ResNet_vd(nn.Layer):
|
||||
def __init__(
|
||||
self, in_channels=3, layers=50, dcn_stage=None, out_indices=None, **kwargs
|
||||
):
|
||||
super(ResNet_vd, self).__init__()
|
||||
|
||||
self.layers = layers
|
||||
supported_layers = [18, 34, 50, 101, 152, 200]
|
||||
assert (
|
||||
layers in supported_layers
|
||||
), "supported layers are {} but input layer is {}".format(
|
||||
supported_layers, layers
|
||||
)
|
||||
|
||||
if layers == 18:
|
||||
depth = [2, 2, 2, 2]
|
||||
elif layers == 34 or layers == 50:
|
||||
depth = [3, 4, 6, 3]
|
||||
elif layers == 101:
|
||||
depth = [3, 4, 23, 3]
|
||||
elif layers == 152:
|
||||
depth = [3, 8, 36, 3]
|
||||
elif layers == 200:
|
||||
depth = [3, 12, 48, 3]
|
||||
num_channels = [64, 256, 512, 1024] if layers >= 50 else [64, 64, 128, 256]
|
||||
num_filters = [64, 128, 256, 512]
|
||||
|
||||
self.dcn_stage = (
|
||||
dcn_stage if dcn_stage is not None else [False, False, False, False]
|
||||
)
|
||||
self.out_indices = out_indices if out_indices is not None else [0, 1, 2, 3]
|
||||
|
||||
self.conv1_1 = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=32,
|
||||
kernel_size=3,
|
||||
stride=2,
|
||||
act="relu",
|
||||
)
|
||||
self.conv1_2 = ConvBNLayer(
|
||||
in_channels=32, out_channels=32, kernel_size=3, stride=1, act="relu"
|
||||
)
|
||||
self.conv1_3 = ConvBNLayer(
|
||||
in_channels=32, out_channels=64, kernel_size=3, stride=1, act="relu"
|
||||
)
|
||||
self.pool2d_max = nn.MaxPool2D(kernel_size=3, stride=2, padding=1)
|
||||
|
||||
self.stages = []
|
||||
self.out_channels = []
|
||||
if layers >= 50:
|
||||
for block in range(len(depth)):
|
||||
block_list = []
|
||||
shortcut = False
|
||||
is_dcn = self.dcn_stage[block]
|
||||
for i in range(depth[block]):
|
||||
bottleneck_block = self.add_sublayer(
|
||||
"bb_%d_%d" % (block, i),
|
||||
BottleneckBlock(
|
||||
in_channels=(
|
||||
num_channels[block]
|
||||
if i == 0
|
||||
else num_filters[block] * 4
|
||||
),
|
||||
out_channels=num_filters[block],
|
||||
stride=2 if i == 0 and block != 0 else 1,
|
||||
shortcut=shortcut,
|
||||
if_first=block == i == 0,
|
||||
is_dcn=is_dcn,
|
||||
),
|
||||
)
|
||||
shortcut = True
|
||||
block_list.append(bottleneck_block)
|
||||
if block in self.out_indices:
|
||||
self.out_channels.append(num_filters[block] * 4)
|
||||
self.stages.append(nn.Sequential(*block_list))
|
||||
else:
|
||||
for block in range(len(depth)):
|
||||
block_list = []
|
||||
shortcut = False
|
||||
for i in range(depth[block]):
|
||||
basic_block = self.add_sublayer(
|
||||
"bb_%d_%d" % (block, i),
|
||||
BasicBlock(
|
||||
in_channels=(
|
||||
num_channels[block] if i == 0 else num_filters[block]
|
||||
),
|
||||
out_channels=num_filters[block],
|
||||
stride=2 if i == 0 and block != 0 else 1,
|
||||
shortcut=shortcut,
|
||||
if_first=block == i == 0,
|
||||
),
|
||||
)
|
||||
shortcut = True
|
||||
block_list.append(basic_block)
|
||||
if block in self.out_indices:
|
||||
self.out_channels.append(num_filters[block])
|
||||
self.stages.append(nn.Sequential(*block_list))
|
||||
|
||||
def forward(self, inputs):
|
||||
y = self.conv1_1(inputs)
|
||||
y = self.conv1_2(y)
|
||||
y = self.conv1_3(y)
|
||||
y = self.pool2d_max(y)
|
||||
out = []
|
||||
for i, block in enumerate(self.stages):
|
||||
y = block(y)
|
||||
if i in self.out_indices:
|
||||
out.append(y)
|
||||
return out
|
||||
314
ppocr/modeling/backbones/det_resnet_vd_sast.py
Normal file
314
ppocr/modeling/backbones/det_resnet_vd_sast.py
Normal file
@@ -0,0 +1,314 @@
|
||||
# copyright (c) 2020 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import ParamAttr
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
|
||||
__all__ = ["ResNet_SAST"]
|
||||
|
||||
|
||||
class ConvBNLayer(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
kernel_size,
|
||||
stride=1,
|
||||
groups=1,
|
||||
is_vd_mode=False,
|
||||
act=None,
|
||||
name=None,
|
||||
):
|
||||
super(ConvBNLayer, self).__init__()
|
||||
|
||||
self.is_vd_mode = is_vd_mode
|
||||
self._pool2d_avg = nn.AvgPool2D(
|
||||
kernel_size=2, stride=2, padding=0, ceil_mode=True
|
||||
)
|
||||
self._conv = nn.Conv2D(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=kernel_size,
|
||||
stride=stride,
|
||||
padding=(kernel_size - 1) // 2,
|
||||
groups=groups,
|
||||
weight_attr=ParamAttr(name=name + "_weights"),
|
||||
bias_attr=False,
|
||||
)
|
||||
if name == "conv1":
|
||||
bn_name = "bn_" + name
|
||||
else:
|
||||
bn_name = "bn" + name[3:]
|
||||
self._batch_norm = nn.BatchNorm(
|
||||
out_channels,
|
||||
act=act,
|
||||
param_attr=ParamAttr(name=bn_name + "_scale"),
|
||||
bias_attr=ParamAttr(bn_name + "_offset"),
|
||||
moving_mean_name=bn_name + "_mean",
|
||||
moving_variance_name=bn_name + "_variance",
|
||||
)
|
||||
|
||||
def forward(self, inputs):
|
||||
if self.is_vd_mode:
|
||||
inputs = self._pool2d_avg(inputs)
|
||||
y = self._conv(inputs)
|
||||
y = self._batch_norm(y)
|
||||
return y
|
||||
|
||||
|
||||
class BottleneckBlock(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
stride,
|
||||
shortcut=True,
|
||||
if_first=False,
|
||||
name=None,
|
||||
):
|
||||
super(BottleneckBlock, self).__init__()
|
||||
|
||||
self.conv0 = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=1,
|
||||
act="relu",
|
||||
name=name + "_branch2a",
|
||||
)
|
||||
self.conv1 = ConvBNLayer(
|
||||
in_channels=out_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
act="relu",
|
||||
name=name + "_branch2b",
|
||||
)
|
||||
self.conv2 = ConvBNLayer(
|
||||
in_channels=out_channels,
|
||||
out_channels=out_channels * 4,
|
||||
kernel_size=1,
|
||||
act=None,
|
||||
name=name + "_branch2c",
|
||||
)
|
||||
|
||||
if not shortcut:
|
||||
self.short = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels * 4,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
is_vd_mode=False if if_first else True,
|
||||
name=name + "_branch1",
|
||||
)
|
||||
|
||||
self.shortcut = shortcut
|
||||
|
||||
def forward(self, inputs):
|
||||
y = self.conv0(inputs)
|
||||
conv1 = self.conv1(y)
|
||||
conv2 = self.conv2(conv1)
|
||||
|
||||
if self.shortcut:
|
||||
short = inputs
|
||||
else:
|
||||
short = self.short(inputs)
|
||||
y = paddle.add(x=short, y=conv2)
|
||||
y = F.relu(y)
|
||||
return y
|
||||
|
||||
|
||||
class BasicBlock(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
stride,
|
||||
shortcut=True,
|
||||
if_first=False,
|
||||
name=None,
|
||||
):
|
||||
super(BasicBlock, self).__init__()
|
||||
self.stride = stride
|
||||
self.conv0 = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
act="relu",
|
||||
name=name + "_branch2a",
|
||||
)
|
||||
self.conv1 = ConvBNLayer(
|
||||
in_channels=out_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=3,
|
||||
act=None,
|
||||
name=name + "_branch2b",
|
||||
)
|
||||
|
||||
if not shortcut:
|
||||
self.short = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
is_vd_mode=False if if_first else True,
|
||||
name=name + "_branch1",
|
||||
)
|
||||
|
||||
self.shortcut = shortcut
|
||||
|
||||
def forward(self, inputs):
|
||||
y = self.conv0(inputs)
|
||||
conv1 = self.conv1(y)
|
||||
|
||||
if self.shortcut:
|
||||
short = inputs
|
||||
else:
|
||||
short = self.short(inputs)
|
||||
y = paddle.add(x=short, y=conv1)
|
||||
y = F.relu(y)
|
||||
return y
|
||||
|
||||
|
||||
class ResNet_SAST(nn.Layer):
|
||||
def __init__(self, in_channels=3, layers=50, **kwargs):
|
||||
super(ResNet_SAST, self).__init__()
|
||||
|
||||
self.layers = layers
|
||||
supported_layers = [18, 34, 50, 101, 152, 200]
|
||||
assert (
|
||||
layers in supported_layers
|
||||
), "supported layers are {} but input layer is {}".format(
|
||||
supported_layers, layers
|
||||
)
|
||||
|
||||
if layers == 18:
|
||||
depth = [2, 2, 2, 2]
|
||||
elif layers == 34 or layers == 50:
|
||||
# depth = [3, 4, 6, 3]
|
||||
depth = [3, 4, 6, 3, 3]
|
||||
elif layers == 101:
|
||||
depth = [3, 4, 23, 3]
|
||||
elif layers == 152:
|
||||
depth = [3, 8, 36, 3]
|
||||
elif layers == 200:
|
||||
depth = [3, 12, 48, 3]
|
||||
# num_channels = [64, 256, 512,
|
||||
# 1024] if layers >= 50 else [64, 64, 128, 256]
|
||||
# num_filters = [64, 128, 256, 512]
|
||||
num_channels = (
|
||||
[64, 256, 512, 1024, 2048] if layers >= 50 else [64, 64, 128, 256]
|
||||
)
|
||||
num_filters = [64, 128, 256, 512, 512]
|
||||
|
||||
self.conv1_1 = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=32,
|
||||
kernel_size=3,
|
||||
stride=2,
|
||||
act="relu",
|
||||
name="conv1_1",
|
||||
)
|
||||
self.conv1_2 = ConvBNLayer(
|
||||
in_channels=32,
|
||||
out_channels=32,
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
act="relu",
|
||||
name="conv1_2",
|
||||
)
|
||||
self.conv1_3 = ConvBNLayer(
|
||||
in_channels=32,
|
||||
out_channels=64,
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
act="relu",
|
||||
name="conv1_3",
|
||||
)
|
||||
self.pool2d_max = nn.MaxPool2D(kernel_size=3, stride=2, padding=1)
|
||||
|
||||
self.stages = []
|
||||
self.out_channels = [3, 64]
|
||||
if layers >= 50:
|
||||
for block in range(len(depth)):
|
||||
block_list = []
|
||||
shortcut = False
|
||||
for i in range(depth[block]):
|
||||
if layers in [101, 152] and block == 2:
|
||||
if i == 0:
|
||||
conv_name = "res" + str(block + 2) + "a"
|
||||
else:
|
||||
conv_name = "res" + str(block + 2) + "b" + str(i)
|
||||
else:
|
||||
conv_name = "res" + str(block + 2) + chr(97 + i)
|
||||
bottleneck_block = self.add_sublayer(
|
||||
"bb_%d_%d" % (block, i),
|
||||
BottleneckBlock(
|
||||
in_channels=(
|
||||
num_channels[block]
|
||||
if i == 0
|
||||
else num_filters[block] * 4
|
||||
),
|
||||
out_channels=num_filters[block],
|
||||
stride=2 if i == 0 and block != 0 else 1,
|
||||
shortcut=shortcut,
|
||||
if_first=block == i == 0,
|
||||
name=conv_name,
|
||||
),
|
||||
)
|
||||
shortcut = True
|
||||
block_list.append(bottleneck_block)
|
||||
self.out_channels.append(num_filters[block] * 4)
|
||||
self.stages.append(nn.Sequential(*block_list))
|
||||
else:
|
||||
for block in range(len(depth)):
|
||||
block_list = []
|
||||
shortcut = False
|
||||
for i in range(depth[block]):
|
||||
conv_name = "res" + str(block + 2) + chr(97 + i)
|
||||
basic_block = self.add_sublayer(
|
||||
"bb_%d_%d" % (block, i),
|
||||
BasicBlock(
|
||||
in_channels=(
|
||||
num_channels[block] if i == 0 else num_filters[block]
|
||||
),
|
||||
out_channels=num_filters[block],
|
||||
stride=2 if i == 0 and block != 0 else 1,
|
||||
shortcut=shortcut,
|
||||
if_first=block == i == 0,
|
||||
name=conv_name,
|
||||
),
|
||||
)
|
||||
shortcut = True
|
||||
block_list.append(basic_block)
|
||||
self.out_channels.append(num_filters[block])
|
||||
self.stages.append(nn.Sequential(*block_list))
|
||||
|
||||
def forward(self, inputs):
|
||||
out = [inputs]
|
||||
y = self.conv1_1(inputs)
|
||||
y = self.conv1_2(y)
|
||||
y = self.conv1_3(y)
|
||||
out.append(y)
|
||||
y = self.pool2d_max(y)
|
||||
for block in self.stages:
|
||||
y = block(y)
|
||||
out.append(y)
|
||||
return out
|
||||
292
ppocr/modeling/backbones/e2e_resnet_vd_pg.py
Normal file
292
ppocr/modeling/backbones/e2e_resnet_vd_pg.py
Normal file
@@ -0,0 +1,292 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import ParamAttr
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
|
||||
__all__ = ["ResNet"]
|
||||
|
||||
|
||||
class ConvBNLayer(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
kernel_size,
|
||||
stride=1,
|
||||
groups=1,
|
||||
is_vd_mode=False,
|
||||
act=None,
|
||||
name=None,
|
||||
):
|
||||
super(ConvBNLayer, self).__init__()
|
||||
|
||||
self.is_vd_mode = is_vd_mode
|
||||
self._pool2d_avg = nn.AvgPool2D(
|
||||
kernel_size=2, stride=2, padding=0, ceil_mode=True
|
||||
)
|
||||
self._conv = nn.Conv2D(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=kernel_size,
|
||||
stride=stride,
|
||||
padding=(kernel_size - 1) // 2,
|
||||
groups=groups,
|
||||
weight_attr=ParamAttr(name=name + "_weights"),
|
||||
bias_attr=False,
|
||||
)
|
||||
if name == "conv1":
|
||||
bn_name = "bn_" + name
|
||||
else:
|
||||
bn_name = "bn" + name[3:]
|
||||
self._batch_norm = nn.BatchNorm(
|
||||
out_channels,
|
||||
act=act,
|
||||
param_attr=ParamAttr(name=bn_name + "_scale"),
|
||||
bias_attr=ParamAttr(bn_name + "_offset"),
|
||||
moving_mean_name=bn_name + "_mean",
|
||||
moving_variance_name=bn_name + "_variance",
|
||||
)
|
||||
|
||||
def forward(self, inputs):
|
||||
y = self._conv(inputs)
|
||||
y = self._batch_norm(y)
|
||||
return y
|
||||
|
||||
|
||||
class BottleneckBlock(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
stride,
|
||||
shortcut=True,
|
||||
if_first=False,
|
||||
name=None,
|
||||
):
|
||||
super(BottleneckBlock, self).__init__()
|
||||
|
||||
self.conv0 = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=1,
|
||||
act="relu",
|
||||
name=name + "_branch2a",
|
||||
)
|
||||
self.conv1 = ConvBNLayer(
|
||||
in_channels=out_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
act="relu",
|
||||
name=name + "_branch2b",
|
||||
)
|
||||
self.conv2 = ConvBNLayer(
|
||||
in_channels=out_channels,
|
||||
out_channels=out_channels * 4,
|
||||
kernel_size=1,
|
||||
act=None,
|
||||
name=name + "_branch2c",
|
||||
)
|
||||
|
||||
if not shortcut:
|
||||
self.short = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels * 4,
|
||||
kernel_size=1,
|
||||
stride=stride,
|
||||
is_vd_mode=False if if_first else True,
|
||||
name=name + "_branch1",
|
||||
)
|
||||
|
||||
self.shortcut = shortcut
|
||||
|
||||
def forward(self, inputs):
|
||||
y = self.conv0(inputs)
|
||||
conv1 = self.conv1(y)
|
||||
conv2 = self.conv2(conv1)
|
||||
|
||||
if self.shortcut:
|
||||
short = inputs
|
||||
else:
|
||||
short = self.short(inputs)
|
||||
y = paddle.add(x=short, y=conv2)
|
||||
y = F.relu(y)
|
||||
return y
|
||||
|
||||
|
||||
class BasicBlock(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
stride,
|
||||
shortcut=True,
|
||||
if_first=False,
|
||||
name=None,
|
||||
):
|
||||
super(BasicBlock, self).__init__()
|
||||
self.stride = stride
|
||||
self.conv0 = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
act="relu",
|
||||
name=name + "_branch2a",
|
||||
)
|
||||
self.conv1 = ConvBNLayer(
|
||||
in_channels=out_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=3,
|
||||
act=None,
|
||||
name=name + "_branch2b",
|
||||
)
|
||||
|
||||
if not shortcut:
|
||||
self.short = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
is_vd_mode=False if if_first else True,
|
||||
name=name + "_branch1",
|
||||
)
|
||||
|
||||
self.shortcut = shortcut
|
||||
|
||||
def forward(self, inputs):
|
||||
y = self.conv0(inputs)
|
||||
conv1 = self.conv1(y)
|
||||
|
||||
if self.shortcut:
|
||||
short = inputs
|
||||
else:
|
||||
short = self.short(inputs)
|
||||
y = paddle.add(x=short, y=conv1)
|
||||
y = F.relu(y)
|
||||
return y
|
||||
|
||||
|
||||
class ResNet(nn.Layer):
|
||||
def __init__(self, in_channels=3, layers=50, **kwargs):
|
||||
super(ResNet, self).__init__()
|
||||
|
||||
self.layers = layers
|
||||
supported_layers = [18, 34, 50, 101, 152, 200]
|
||||
assert (
|
||||
layers in supported_layers
|
||||
), "supported layers are {} but input layer is {}".format(
|
||||
supported_layers, layers
|
||||
)
|
||||
|
||||
if layers == 18:
|
||||
depth = [2, 2, 2, 2]
|
||||
elif layers == 34 or layers == 50:
|
||||
# depth = [3, 4, 6, 3]
|
||||
depth = [3, 4, 6, 3, 3]
|
||||
elif layers == 101:
|
||||
depth = [3, 4, 23, 3]
|
||||
elif layers == 152:
|
||||
depth = [3, 8, 36, 3]
|
||||
elif layers == 200:
|
||||
depth = [3, 12, 48, 3]
|
||||
num_channels = (
|
||||
[64, 256, 512, 1024, 2048] if layers >= 50 else [64, 64, 128, 256]
|
||||
)
|
||||
num_filters = [64, 128, 256, 512, 512]
|
||||
|
||||
self.conv1_1 = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=64,
|
||||
kernel_size=7,
|
||||
stride=2,
|
||||
act="relu",
|
||||
name="conv1_1",
|
||||
)
|
||||
self.pool2d_max = nn.MaxPool2D(kernel_size=3, stride=2, padding=1)
|
||||
|
||||
self.stages = []
|
||||
self.out_channels = [3, 64]
|
||||
# num_filters = [64, 128, 256, 512, 512]
|
||||
if layers >= 50:
|
||||
for block in range(len(depth)):
|
||||
block_list = []
|
||||
shortcut = False
|
||||
for i in range(depth[block]):
|
||||
if layers in [101, 152] and block == 2:
|
||||
if i == 0:
|
||||
conv_name = "res" + str(block + 2) + "a"
|
||||
else:
|
||||
conv_name = "res" + str(block + 2) + "b" + str(i)
|
||||
else:
|
||||
conv_name = "res" + str(block + 2) + chr(97 + i)
|
||||
bottleneck_block = self.add_sublayer(
|
||||
"bb_%d_%d" % (block, i),
|
||||
BottleneckBlock(
|
||||
in_channels=(
|
||||
num_channels[block]
|
||||
if i == 0
|
||||
else num_filters[block] * 4
|
||||
),
|
||||
out_channels=num_filters[block],
|
||||
stride=2 if i == 0 and block != 0 else 1,
|
||||
shortcut=shortcut,
|
||||
if_first=block == i == 0,
|
||||
name=conv_name,
|
||||
),
|
||||
)
|
||||
shortcut = True
|
||||
block_list.append(bottleneck_block)
|
||||
self.out_channels.append(num_filters[block] * 4)
|
||||
self.stages.append(nn.Sequential(*block_list))
|
||||
else:
|
||||
for block in range(len(depth)):
|
||||
block_list = []
|
||||
shortcut = False
|
||||
for i in range(depth[block]):
|
||||
conv_name = "res" + str(block + 2) + chr(97 + i)
|
||||
basic_block = self.add_sublayer(
|
||||
"bb_%d_%d" % (block, i),
|
||||
BasicBlock(
|
||||
in_channels=(
|
||||
num_channels[block] if i == 0 else num_filters[block]
|
||||
),
|
||||
out_channels=num_filters[block],
|
||||
stride=2 if i == 0 and block != 0 else 1,
|
||||
shortcut=shortcut,
|
||||
if_first=block == i == 0,
|
||||
name=conv_name,
|
||||
),
|
||||
)
|
||||
shortcut = True
|
||||
block_list.append(basic_block)
|
||||
self.out_channels.append(num_filters[block])
|
||||
self.stages.append(nn.Sequential(*block_list))
|
||||
|
||||
def forward(self, inputs):
|
||||
out = [inputs]
|
||||
y = self.conv1_1(inputs)
|
||||
out.append(y)
|
||||
y = self.pool2d_max(y)
|
||||
for block in self.stages:
|
||||
y = block(y)
|
||||
out.append(y)
|
||||
return out
|
||||
199
ppocr/modeling/backbones/kie_unet_sdmgr.py
Normal file
199
ppocr/modeling/backbones/kie_unet_sdmgr.py
Normal file
@@ -0,0 +1,199 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
import numpy as np
|
||||
import cv2
|
||||
|
||||
__all__ = ["Kie_backbone"]
|
||||
|
||||
|
||||
class Encoder(nn.Layer):
|
||||
def __init__(self, num_channels, num_filters):
|
||||
super(Encoder, self).__init__()
|
||||
self.conv1 = nn.Conv2D(
|
||||
num_channels,
|
||||
num_filters,
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn1 = nn.BatchNorm(num_filters, act="relu")
|
||||
|
||||
self.conv2 = nn.Conv2D(
|
||||
num_filters,
|
||||
num_filters,
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn2 = nn.BatchNorm(num_filters, act="relu")
|
||||
|
||||
self.pool = nn.MaxPool2D(kernel_size=3, stride=2, padding=1)
|
||||
|
||||
def forward(self, inputs):
|
||||
x = self.conv1(inputs)
|
||||
x = self.bn1(x)
|
||||
x = self.conv2(x)
|
||||
x = self.bn2(x)
|
||||
x_pooled = self.pool(x)
|
||||
return x, x_pooled
|
||||
|
||||
|
||||
class Decoder(nn.Layer):
|
||||
def __init__(self, num_channels, num_filters):
|
||||
super(Decoder, self).__init__()
|
||||
|
||||
self.conv1 = nn.Conv2D(
|
||||
num_channels,
|
||||
num_filters,
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn1 = nn.BatchNorm(num_filters, act="relu")
|
||||
|
||||
self.conv2 = nn.Conv2D(
|
||||
num_filters,
|
||||
num_filters,
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn2 = nn.BatchNorm(num_filters, act="relu")
|
||||
|
||||
self.conv0 = nn.Conv2D(
|
||||
num_channels,
|
||||
num_filters,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
padding=0,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn0 = nn.BatchNorm(num_filters, act="relu")
|
||||
|
||||
def forward(self, inputs_prev, inputs):
|
||||
x = self.conv0(inputs)
|
||||
x = self.bn0(x)
|
||||
x = paddle.nn.functional.interpolate(
|
||||
x, scale_factor=2, mode="bilinear", align_corners=False
|
||||
)
|
||||
x = paddle.concat([inputs_prev, x], axis=1)
|
||||
x = self.conv1(x)
|
||||
x = self.bn1(x)
|
||||
x = self.conv2(x)
|
||||
x = self.bn2(x)
|
||||
return x
|
||||
|
||||
|
||||
class UNet(nn.Layer):
|
||||
def __init__(self):
|
||||
super(UNet, self).__init__()
|
||||
self.down1 = Encoder(num_channels=3, num_filters=16)
|
||||
self.down2 = Encoder(num_channels=16, num_filters=32)
|
||||
self.down3 = Encoder(num_channels=32, num_filters=64)
|
||||
self.down4 = Encoder(num_channels=64, num_filters=128)
|
||||
self.down5 = Encoder(num_channels=128, num_filters=256)
|
||||
|
||||
self.up1 = Decoder(32, 16)
|
||||
self.up2 = Decoder(64, 32)
|
||||
self.up3 = Decoder(128, 64)
|
||||
self.up4 = Decoder(256, 128)
|
||||
self.out_channels = 16
|
||||
|
||||
def forward(self, inputs):
|
||||
x1, _ = self.down1(inputs)
|
||||
_, x2 = self.down2(x1)
|
||||
_, x3 = self.down3(x2)
|
||||
_, x4 = self.down4(x3)
|
||||
_, x5 = self.down5(x4)
|
||||
|
||||
x = self.up4(x4, x5)
|
||||
x = self.up3(x3, x)
|
||||
x = self.up2(x2, x)
|
||||
x = self.up1(x1, x)
|
||||
return x
|
||||
|
||||
|
||||
class Kie_backbone(nn.Layer):
|
||||
def __init__(self, in_channels, **kwargs):
|
||||
super(Kie_backbone, self).__init__()
|
||||
self.out_channels = 16
|
||||
self.img_feat = UNet()
|
||||
self.maxpool = nn.MaxPool2D(kernel_size=7)
|
||||
|
||||
def bbox2roi(self, bbox_list):
|
||||
rois_list = []
|
||||
rois_num = []
|
||||
for img_id, bboxes in enumerate(bbox_list):
|
||||
rois_num.append(bboxes.shape[0])
|
||||
rois_list.append(bboxes)
|
||||
rois = paddle.concat(rois_list, 0)
|
||||
rois_num = paddle.to_tensor(rois_num, dtype="int32")
|
||||
return rois, rois_num
|
||||
|
||||
def pre_process(self, img, relations, texts, gt_bboxes, tag, img_size):
|
||||
img, relations, texts, gt_bboxes, tag, img_size = (
|
||||
img.numpy(),
|
||||
relations.numpy(),
|
||||
texts.numpy(),
|
||||
gt_bboxes.numpy(),
|
||||
tag.numpy().tolist(),
|
||||
img_size.numpy(),
|
||||
)
|
||||
temp_relations, temp_texts, temp_gt_bboxes = [], [], []
|
||||
h, w = int(np.max(img_size[:, 0])), int(np.max(img_size[:, 1]))
|
||||
img = paddle.to_tensor(img[:, :, :h, :w])
|
||||
batch = len(tag)
|
||||
for i in range(batch):
|
||||
num, recoder_len = tag[i][0], tag[i][1]
|
||||
temp_relations.append(
|
||||
paddle.to_tensor(relations[i, :num, :num, :], dtype="float32")
|
||||
)
|
||||
temp_texts.append(
|
||||
paddle.to_tensor(texts[i, :num, :recoder_len], dtype="float32")
|
||||
)
|
||||
temp_gt_bboxes.append(
|
||||
paddle.to_tensor(gt_bboxes[i, :num, ...], dtype="float32")
|
||||
)
|
||||
return img, temp_relations, temp_texts, temp_gt_bboxes
|
||||
|
||||
def forward(self, inputs):
|
||||
img = inputs[0]
|
||||
relations, texts, gt_bboxes, tag, img_size = (
|
||||
inputs[1],
|
||||
inputs[2],
|
||||
inputs[3],
|
||||
inputs[5],
|
||||
inputs[-1],
|
||||
)
|
||||
img, relations, texts, gt_bboxes = self.pre_process(
|
||||
img, relations, texts, gt_bboxes, tag, img_size
|
||||
)
|
||||
x = self.img_feat(img)
|
||||
boxes, rois_num = self.bbox2roi(gt_bboxes)
|
||||
feats = paddle.vision.ops.roi_align(
|
||||
x, boxes, spatial_scale=1.0, output_size=7, boxes_num=rois_num
|
||||
)
|
||||
feats = self.maxpool(feats).squeeze(-1).squeeze(-1)
|
||||
return [relations, texts, feats]
|
||||
150
ppocr/modeling/backbones/rec_densenet.py
Normal file
150
ppocr/modeling/backbones/rec_densenet.py
Normal file
@@ -0,0 +1,150 @@
|
||||
# copyright (c) 2020 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/LBH1024/CAN/models/densenet.py
|
||||
|
||||
"""
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import math
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
|
||||
|
||||
class Bottleneck(nn.Layer):
|
||||
def __init__(self, nChannels, growthRate, use_dropout):
|
||||
super(Bottleneck, self).__init__()
|
||||
interChannels = 4 * growthRate
|
||||
self.bn1 = nn.BatchNorm2D(interChannels)
|
||||
self.conv1 = nn.Conv2D(
|
||||
nChannels, interChannels, kernel_size=1, bias_attr=None
|
||||
) # Xavier initialization
|
||||
self.bn2 = nn.BatchNorm2D(growthRate)
|
||||
self.conv2 = nn.Conv2D(
|
||||
interChannels, growthRate, kernel_size=3, padding=1, bias_attr=None
|
||||
) # Xavier initialization
|
||||
self.use_dropout = use_dropout
|
||||
self.dropout = nn.Dropout(p=0.2)
|
||||
|
||||
def forward(self, x):
|
||||
out = F.relu(self.bn1(self.conv1(x)))
|
||||
if self.use_dropout:
|
||||
out = self.dropout(out)
|
||||
out = F.relu(self.bn2(self.conv2(out)))
|
||||
if self.use_dropout:
|
||||
out = self.dropout(out)
|
||||
out = paddle.concat([x, out], 1)
|
||||
return out
|
||||
|
||||
|
||||
class SingleLayer(nn.Layer):
|
||||
def __init__(self, nChannels, growthRate, use_dropout):
|
||||
super(SingleLayer, self).__init__()
|
||||
self.bn1 = nn.BatchNorm2D(nChannels)
|
||||
self.conv1 = nn.Conv2D(
|
||||
nChannels, growthRate, kernel_size=3, padding=1, bias_attr=False
|
||||
)
|
||||
|
||||
self.use_dropout = use_dropout
|
||||
self.dropout = nn.Dropout(p=0.2)
|
||||
|
||||
def forward(self, x):
|
||||
out = self.conv1(F.relu(x))
|
||||
if self.use_dropout:
|
||||
out = self.dropout(out)
|
||||
|
||||
out = paddle.concat([x, out], 1)
|
||||
return out
|
||||
|
||||
|
||||
class Transition(nn.Layer):
|
||||
def __init__(self, nChannels, out_channels, use_dropout):
|
||||
super(Transition, self).__init__()
|
||||
self.bn1 = nn.BatchNorm2D(out_channels)
|
||||
self.conv1 = nn.Conv2D(nChannels, out_channels, kernel_size=1, bias_attr=False)
|
||||
self.use_dropout = use_dropout
|
||||
self.dropout = nn.Dropout(p=0.2)
|
||||
|
||||
def forward(self, x):
|
||||
out = F.relu(self.bn1(self.conv1(x)))
|
||||
if self.use_dropout:
|
||||
out = self.dropout(out)
|
||||
out = F.avg_pool2d(out, 2, ceil_mode=True, exclusive=False)
|
||||
return out
|
||||
|
||||
|
||||
class DenseNet(nn.Layer):
|
||||
def __init__(
|
||||
self, growthRate, reduction, bottleneck, use_dropout, input_channel, **kwargs
|
||||
):
|
||||
super(DenseNet, self).__init__()
|
||||
|
||||
nDenseBlocks = 16
|
||||
nChannels = 2 * growthRate
|
||||
|
||||
self.conv1 = nn.Conv2D(
|
||||
input_channel,
|
||||
nChannels,
|
||||
kernel_size=7,
|
||||
padding=3,
|
||||
stride=2,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.dense1 = self._make_dense(
|
||||
nChannels, growthRate, nDenseBlocks, bottleneck, use_dropout
|
||||
)
|
||||
nChannels += nDenseBlocks * growthRate
|
||||
out_channels = int(math.floor(nChannels * reduction))
|
||||
self.trans1 = Transition(nChannels, out_channels, use_dropout)
|
||||
|
||||
nChannels = out_channels
|
||||
self.dense2 = self._make_dense(
|
||||
nChannels, growthRate, nDenseBlocks, bottleneck, use_dropout
|
||||
)
|
||||
nChannels += nDenseBlocks * growthRate
|
||||
out_channels = int(math.floor(nChannels * reduction))
|
||||
self.trans2 = Transition(nChannels, out_channels, use_dropout)
|
||||
|
||||
nChannels = out_channels
|
||||
self.dense3 = self._make_dense(
|
||||
nChannels, growthRate, nDenseBlocks, bottleneck, use_dropout
|
||||
)
|
||||
self.out_channels = out_channels
|
||||
|
||||
def _make_dense(self, nChannels, growthRate, nDenseBlocks, bottleneck, use_dropout):
|
||||
layers = []
|
||||
for i in range(int(nDenseBlocks)):
|
||||
if bottleneck:
|
||||
layers.append(Bottleneck(nChannels, growthRate, use_dropout))
|
||||
else:
|
||||
layers.append(SingleLayer(nChannels, growthRate, use_dropout))
|
||||
nChannels += growthRate
|
||||
return nn.Sequential(*layers)
|
||||
|
||||
def forward(self, inputs):
|
||||
x, x_m, y = inputs
|
||||
out = self.conv1(x)
|
||||
out = F.relu(out)
|
||||
out = F.max_pool2d(out, 2, ceil_mode=True)
|
||||
out = self.dense1(out)
|
||||
out = self.trans1(out)
|
||||
out = self.dense2(out)
|
||||
out = self.trans2(out)
|
||||
out = self.dense3(out)
|
||||
return out, x_m, y
|
||||
1296
ppocr/modeling/backbones/rec_donut_swin.py
Normal file
1296
ppocr/modeling/backbones/rec_donut_swin.py
Normal file
File diff suppressed because it is too large
Load Diff
305
ppocr/modeling/backbones/rec_efficientb3_pren.py
Normal file
305
ppocr/modeling/backbones/rec_efficientb3_pren.py
Normal file
@@ -0,0 +1,305 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
Code is refer from:
|
||||
https://github.com/RuijieJ/pren/blob/main/Nets/EfficientNet.py
|
||||
"""
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import math
|
||||
import re
|
||||
import collections
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
|
||||
__all__ = ["EfficientNetb3_PREN"]
|
||||
|
||||
GlobalParams = collections.namedtuple(
|
||||
"GlobalParams",
|
||||
[
|
||||
"batch_norm_momentum",
|
||||
"batch_norm_epsilon",
|
||||
"dropout_rate",
|
||||
"num_classes",
|
||||
"width_coefficient",
|
||||
"depth_coefficient",
|
||||
"depth_divisor",
|
||||
"min_depth",
|
||||
"drop_connect_rate",
|
||||
"image_size",
|
||||
],
|
||||
)
|
||||
|
||||
BlockArgs = collections.namedtuple(
|
||||
"BlockArgs",
|
||||
[
|
||||
"kernel_size",
|
||||
"num_repeat",
|
||||
"input_filters",
|
||||
"output_filters",
|
||||
"expand_ratio",
|
||||
"id_skip",
|
||||
"stride",
|
||||
"se_ratio",
|
||||
],
|
||||
)
|
||||
|
||||
|
||||
class BlockDecoder:
|
||||
@staticmethod
|
||||
def _decode_block_string(block_string):
|
||||
assert isinstance(block_string, str)
|
||||
|
||||
ops = block_string.split("_")
|
||||
options = {}
|
||||
for op in ops:
|
||||
splits = re.split(r"(\d.*)", op)
|
||||
if len(splits) >= 2:
|
||||
key, value = splits[:2]
|
||||
options[key] = value
|
||||
|
||||
assert ("s" in options and len(options["s"]) == 1) or (
|
||||
len(options["s"]) == 2 and options["s"][0] == options["s"][1]
|
||||
)
|
||||
|
||||
return BlockArgs(
|
||||
kernel_size=int(options["k"]),
|
||||
num_repeat=int(options["r"]),
|
||||
input_filters=int(options["i"]),
|
||||
output_filters=int(options["o"]),
|
||||
expand_ratio=int(options["e"]),
|
||||
id_skip=("noskip" not in block_string),
|
||||
se_ratio=float(options["se"]) if "se" in options else None,
|
||||
stride=[int(options["s"][0])],
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def decode(string_list):
|
||||
assert isinstance(string_list, list)
|
||||
blocks_args = []
|
||||
for block_string in string_list:
|
||||
blocks_args.append(BlockDecoder._decode_block_string(block_string))
|
||||
return blocks_args
|
||||
|
||||
|
||||
def efficientnet(
|
||||
width_coefficient=None,
|
||||
depth_coefficient=None,
|
||||
dropout_rate=0.2,
|
||||
drop_connect_rate=0.2,
|
||||
image_size=None,
|
||||
num_classes=1000,
|
||||
):
|
||||
blocks_args = [
|
||||
"r1_k3_s11_e1_i32_o16_se0.25",
|
||||
"r2_k3_s22_e6_i16_o24_se0.25",
|
||||
"r2_k5_s22_e6_i24_o40_se0.25",
|
||||
"r3_k3_s22_e6_i40_o80_se0.25",
|
||||
"r3_k5_s11_e6_i80_o112_se0.25",
|
||||
"r4_k5_s22_e6_i112_o192_se0.25",
|
||||
"r1_k3_s11_e6_i192_o320_se0.25",
|
||||
]
|
||||
blocks_args = BlockDecoder.decode(blocks_args)
|
||||
|
||||
global_params = GlobalParams(
|
||||
batch_norm_momentum=0.99,
|
||||
batch_norm_epsilon=1e-3,
|
||||
dropout_rate=dropout_rate,
|
||||
drop_connect_rate=drop_connect_rate,
|
||||
num_classes=num_classes,
|
||||
width_coefficient=width_coefficient,
|
||||
depth_coefficient=depth_coefficient,
|
||||
depth_divisor=8,
|
||||
min_depth=None,
|
||||
image_size=image_size,
|
||||
)
|
||||
return blocks_args, global_params
|
||||
|
||||
|
||||
class EffUtils:
|
||||
@staticmethod
|
||||
def round_filters(filters, global_params):
|
||||
"""Calculate and round number of filters based on depth multiplier."""
|
||||
multiplier = global_params.width_coefficient
|
||||
if not multiplier:
|
||||
return filters
|
||||
divisor = global_params.depth_divisor
|
||||
min_depth = global_params.min_depth
|
||||
filters *= multiplier
|
||||
min_depth = min_depth or divisor
|
||||
new_filters = max(min_depth, int(filters + divisor / 2) // divisor * divisor)
|
||||
if new_filters < 0.9 * filters:
|
||||
new_filters += divisor
|
||||
return int(new_filters)
|
||||
|
||||
@staticmethod
|
||||
def round_repeats(repeats, global_params):
|
||||
"""Round number of filters based on depth multiplier."""
|
||||
multiplier = global_params.depth_coefficient
|
||||
if not multiplier:
|
||||
return repeats
|
||||
return int(math.ceil(multiplier * repeats))
|
||||
|
||||
|
||||
class MbConvBlock(nn.Layer):
|
||||
def __init__(self, block_args):
|
||||
super(MbConvBlock, self).__init__()
|
||||
self._block_args = block_args
|
||||
self.has_se = (self._block_args.se_ratio is not None) and (
|
||||
0 < self._block_args.se_ratio <= 1
|
||||
)
|
||||
self.id_skip = block_args.id_skip
|
||||
|
||||
# expansion phase
|
||||
self.inp = self._block_args.input_filters
|
||||
oup = self._block_args.input_filters * self._block_args.expand_ratio
|
||||
if self._block_args.expand_ratio != 1:
|
||||
self._expand_conv = nn.Conv2D(self.inp, oup, 1, bias_attr=False)
|
||||
self._bn0 = nn.BatchNorm(oup)
|
||||
|
||||
# depthwise conv phase
|
||||
k = self._block_args.kernel_size
|
||||
s = self._block_args.stride
|
||||
if isinstance(s, list):
|
||||
s = s[0]
|
||||
self._depthwise_conv = nn.Conv2D(
|
||||
oup,
|
||||
oup,
|
||||
groups=oup,
|
||||
kernel_size=k,
|
||||
stride=s,
|
||||
padding="same",
|
||||
bias_attr=False,
|
||||
)
|
||||
self._bn1 = nn.BatchNorm(oup)
|
||||
|
||||
# squeeze and excitation layer, if desired
|
||||
if self.has_se:
|
||||
num_squeezed_channels = max(
|
||||
1, int(self._block_args.input_filters * self._block_args.se_ratio)
|
||||
)
|
||||
self._se_reduce = nn.Conv2D(oup, num_squeezed_channels, 1)
|
||||
self._se_expand = nn.Conv2D(num_squeezed_channels, oup, 1)
|
||||
|
||||
# output phase and some util class
|
||||
self.final_oup = self._block_args.output_filters
|
||||
self._project_conv = nn.Conv2D(oup, self.final_oup, 1, bias_attr=False)
|
||||
self._bn2 = nn.BatchNorm(self.final_oup)
|
||||
self._swish = nn.Swish()
|
||||
|
||||
def _drop_connect(self, inputs, p, training):
|
||||
if not training:
|
||||
return inputs
|
||||
batch_size = inputs.shape[0]
|
||||
keep_prob = 1 - p
|
||||
random_tensor = keep_prob
|
||||
random_tensor += paddle.rand([batch_size, 1, 1, 1], dtype=inputs.dtype)
|
||||
random_tensor = paddle.to_tensor(random_tensor, place=inputs.place)
|
||||
binary_tensor = paddle.floor(random_tensor)
|
||||
output = inputs / keep_prob * binary_tensor
|
||||
return output
|
||||
|
||||
def forward(self, inputs, drop_connect_rate=None):
|
||||
# expansion and depthwise conv
|
||||
x = inputs
|
||||
if self._block_args.expand_ratio != 1:
|
||||
x = self._swish(self._bn0(self._expand_conv(inputs)))
|
||||
x = self._swish(self._bn1(self._depthwise_conv(x)))
|
||||
|
||||
# squeeze and excitation
|
||||
if self.has_se:
|
||||
x_squeezed = F.adaptive_avg_pool2d(x, 1)
|
||||
x_squeezed = self._se_expand(self._swish(self._se_reduce(x_squeezed)))
|
||||
x = F.sigmoid(x_squeezed) * x
|
||||
x = self._bn2(self._project_conv(x))
|
||||
|
||||
# skip connection and drop connect
|
||||
if self.id_skip and self._block_args.stride == 1 and self.inp == self.final_oup:
|
||||
if drop_connect_rate:
|
||||
x = self._drop_connect(x, p=drop_connect_rate, training=self.training)
|
||||
x = x + inputs
|
||||
return x
|
||||
|
||||
|
||||
class EfficientNetb3_PREN(nn.Layer):
|
||||
def __init__(self, in_channels):
|
||||
super(EfficientNetb3_PREN, self).__init__()
|
||||
"""
|
||||
the fllowing are efficientnetb3's superparams,
|
||||
they means efficientnetb3 network's width, depth, resolution and
|
||||
dropout respectively, to fit for text recognition task, the resolution
|
||||
here is changed from 300 to 64.
|
||||
"""
|
||||
w, d, s, p = 1.2, 1.4, 64, 0.3
|
||||
self._blocks_args, self._global_params = efficientnet(
|
||||
width_coefficient=w, depth_coefficient=d, dropout_rate=p, image_size=s
|
||||
)
|
||||
self.out_channels = []
|
||||
# stem
|
||||
out_channels = EffUtils.round_filters(32, self._global_params)
|
||||
self._conv_stem = nn.Conv2D(
|
||||
in_channels, out_channels, 3, 2, padding="same", bias_attr=False
|
||||
)
|
||||
self._bn0 = nn.BatchNorm(out_channels)
|
||||
|
||||
# build blocks
|
||||
self._blocks = []
|
||||
# to extract three feature maps for fpn based on efficientnetb3 backbone
|
||||
self._concerned_block_idxes = [7, 17, 25]
|
||||
_concerned_idx = 0
|
||||
for i, block_args in enumerate(self._blocks_args):
|
||||
block_args = block_args._replace(
|
||||
input_filters=EffUtils.round_filters(
|
||||
block_args.input_filters, self._global_params
|
||||
),
|
||||
output_filters=EffUtils.round_filters(
|
||||
block_args.output_filters, self._global_params
|
||||
),
|
||||
num_repeat=EffUtils.round_repeats(
|
||||
block_args.num_repeat, self._global_params
|
||||
),
|
||||
)
|
||||
self._blocks.append(self.add_sublayer(f"{i}-0", MbConvBlock(block_args)))
|
||||
_concerned_idx += 1
|
||||
if _concerned_idx in self._concerned_block_idxes:
|
||||
self.out_channels.append(block_args.output_filters)
|
||||
if block_args.num_repeat > 1:
|
||||
block_args = block_args._replace(
|
||||
input_filters=block_args.output_filters, stride=1
|
||||
)
|
||||
for j in range(block_args.num_repeat - 1):
|
||||
self._blocks.append(
|
||||
self.add_sublayer(f"{i}-{j+1}", MbConvBlock(block_args))
|
||||
)
|
||||
_concerned_idx += 1
|
||||
if _concerned_idx in self._concerned_block_idxes:
|
||||
self.out_channels.append(block_args.output_filters)
|
||||
|
||||
self._swish = nn.Swish()
|
||||
|
||||
def forward(self, inputs):
|
||||
outs = []
|
||||
x = self._swish(self._bn0(self._conv_stem(inputs)))
|
||||
for idx, block in enumerate(self._blocks):
|
||||
drop_connect_rate = self._global_params.drop_connect_rate
|
||||
if drop_connect_rate:
|
||||
drop_connect_rate *= float(idx) / len(self._blocks)
|
||||
x = block(x, drop_connect_rate=drop_connect_rate)
|
||||
if idx in self._concerned_block_idxes:
|
||||
outs.append(x)
|
||||
return outs
|
||||
385
ppocr/modeling/backbones/rec_hgnet.py
Normal file
385
ppocr/modeling/backbones/rec_hgnet.py
Normal file
@@ -0,0 +1,385 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
from paddle.nn.initializer import KaimingNormal, Constant
|
||||
from paddle.nn import Conv2D, BatchNorm2D, ReLU, AdaptiveAvgPool2D, MaxPool2D
|
||||
from paddle.regularizer import L2Decay
|
||||
from paddle import ParamAttr
|
||||
|
||||
kaiming_normal_ = KaimingNormal()
|
||||
zeros_ = Constant(value=0.0)
|
||||
ones_ = Constant(value=1.0)
|
||||
|
||||
|
||||
class MeanPool2D(nn.Layer):
|
||||
def __init__(self, w, h):
|
||||
super().__init__()
|
||||
self.w = w
|
||||
self.h = h
|
||||
|
||||
def forward(self, feat):
|
||||
batch_size, channels, _, _ = feat.shape
|
||||
feat_flat = paddle.reshape(feat, [batch_size, channels, -1])
|
||||
feat_mean = paddle.mean(feat_flat, axis=2)
|
||||
feat_mean = paddle.reshape(feat_mean, [batch_size, channels, self.w, self.h])
|
||||
return feat_mean
|
||||
|
||||
|
||||
class ConvBNAct(nn.Layer):
|
||||
def __init__(
|
||||
self, in_channels, out_channels, kernel_size, stride, groups=1, use_act=True
|
||||
):
|
||||
super().__init__()
|
||||
self.use_act = use_act
|
||||
self.conv = Conv2D(
|
||||
in_channels,
|
||||
out_channels,
|
||||
kernel_size,
|
||||
stride,
|
||||
padding=(kernel_size - 1) // 2,
|
||||
groups=groups,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn = BatchNorm2D(
|
||||
out_channels,
|
||||
weight_attr=ParamAttr(regularizer=L2Decay(0.0)),
|
||||
bias_attr=ParamAttr(regularizer=L2Decay(0.0)),
|
||||
)
|
||||
if self.use_act:
|
||||
self.act = ReLU()
|
||||
|
||||
def forward(self, x):
|
||||
x = self.conv(x)
|
||||
x = self.bn(x)
|
||||
if self.use_act:
|
||||
x = self.act(x)
|
||||
return x
|
||||
|
||||
|
||||
class ESEModule(nn.Layer):
|
||||
def __init__(self, channels):
|
||||
super().__init__()
|
||||
if "npu" in paddle.device.get_device():
|
||||
self.avg_pool = MeanPool2D(1, 1)
|
||||
else:
|
||||
self.avg_pool = AdaptiveAvgPool2D(1)
|
||||
self.conv = Conv2D(
|
||||
in_channels=channels,
|
||||
out_channels=channels,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
padding=0,
|
||||
)
|
||||
self.sigmoid = nn.Sigmoid()
|
||||
|
||||
def forward(self, x):
|
||||
identity = x
|
||||
x = self.avg_pool(x)
|
||||
x = self.conv(x)
|
||||
x = self.sigmoid(x)
|
||||
return paddle.multiply(x=identity, y=x)
|
||||
|
||||
|
||||
class HG_Block(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
mid_channels,
|
||||
out_channels,
|
||||
layer_num,
|
||||
identity=False,
|
||||
):
|
||||
super().__init__()
|
||||
self.identity = identity
|
||||
|
||||
self.layers = nn.LayerList()
|
||||
self.layers.append(
|
||||
ConvBNAct(
|
||||
in_channels=in_channels,
|
||||
out_channels=mid_channels,
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
)
|
||||
)
|
||||
for _ in range(layer_num - 1):
|
||||
self.layers.append(
|
||||
ConvBNAct(
|
||||
in_channels=mid_channels,
|
||||
out_channels=mid_channels,
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
)
|
||||
)
|
||||
|
||||
# feature aggregation
|
||||
total_channels = in_channels + layer_num * mid_channels
|
||||
self.aggregation_conv = ConvBNAct(
|
||||
in_channels=total_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
)
|
||||
self.att = ESEModule(out_channels)
|
||||
|
||||
def forward(self, x):
|
||||
identity = x
|
||||
output = []
|
||||
output.append(x)
|
||||
for layer in self.layers:
|
||||
x = layer(x)
|
||||
output.append(x)
|
||||
x = paddle.concat(output, axis=1)
|
||||
x = self.aggregation_conv(x)
|
||||
x = self.att(x)
|
||||
if self.identity:
|
||||
x += identity
|
||||
return x
|
||||
|
||||
|
||||
class HG_Stage(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
mid_channels,
|
||||
out_channels,
|
||||
block_num,
|
||||
layer_num,
|
||||
downsample=True,
|
||||
stride=[2, 1],
|
||||
):
|
||||
super().__init__()
|
||||
self.downsample = downsample
|
||||
if downsample:
|
||||
self.downsample = ConvBNAct(
|
||||
in_channels=in_channels,
|
||||
out_channels=in_channels,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
groups=in_channels,
|
||||
use_act=False,
|
||||
)
|
||||
|
||||
blocks_list = []
|
||||
blocks_list.append(
|
||||
HG_Block(in_channels, mid_channels, out_channels, layer_num, identity=False)
|
||||
)
|
||||
for _ in range(block_num - 1):
|
||||
blocks_list.append(
|
||||
HG_Block(
|
||||
out_channels, mid_channels, out_channels, layer_num, identity=True
|
||||
)
|
||||
)
|
||||
self.blocks = nn.Sequential(*blocks_list)
|
||||
|
||||
def forward(self, x):
|
||||
if self.downsample:
|
||||
x = self.downsample(x)
|
||||
x = self.blocks(x)
|
||||
return x
|
||||
|
||||
|
||||
class PPHGNet(nn.Layer):
|
||||
"""
|
||||
PPHGNet
|
||||
Args:
|
||||
stem_channels: list. Stem channel list of PPHGNet.
|
||||
stage_config: dict. The configuration of each stage of PPHGNet. such as the number of channels, stride, etc.
|
||||
layer_num: int. Number of layers of HG_Block.
|
||||
use_last_conv: boolean. Whether to use a 1x1 convolutional layer before the classification layer.
|
||||
class_expand: int=2048. Number of channels for the last 1x1 convolutional layer.
|
||||
dropout_prob: float. Parameters of dropout, 0.0 means dropout is not used.
|
||||
class_num: int=1000. The number of classes.
|
||||
Returns:
|
||||
model: nn.Layer. Specific PPHGNet model depends on args.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
stem_channels,
|
||||
stage_config,
|
||||
layer_num,
|
||||
in_channels=3,
|
||||
det=False,
|
||||
out_indices=None,
|
||||
):
|
||||
super().__init__()
|
||||
self.det = det
|
||||
self.out_indices = out_indices if out_indices is not None else [0, 1, 2, 3]
|
||||
|
||||
# stem
|
||||
stem_channels.insert(0, in_channels)
|
||||
self.stem = nn.Sequential(
|
||||
*[
|
||||
ConvBNAct(
|
||||
in_channels=stem_channels[i],
|
||||
out_channels=stem_channels[i + 1],
|
||||
kernel_size=3,
|
||||
stride=2 if i == 0 else 1,
|
||||
)
|
||||
for i in range(len(stem_channels) - 1)
|
||||
]
|
||||
)
|
||||
|
||||
if self.det:
|
||||
self.pool = nn.MaxPool2D(kernel_size=3, stride=2, padding=1)
|
||||
# stages
|
||||
self.stages = nn.LayerList()
|
||||
self.out_channels = []
|
||||
for block_id, k in enumerate(stage_config):
|
||||
(
|
||||
in_channels,
|
||||
mid_channels,
|
||||
out_channels,
|
||||
block_num,
|
||||
downsample,
|
||||
stride,
|
||||
) = stage_config[k]
|
||||
self.stages.append(
|
||||
HG_Stage(
|
||||
in_channels,
|
||||
mid_channels,
|
||||
out_channels,
|
||||
block_num,
|
||||
layer_num,
|
||||
downsample,
|
||||
stride,
|
||||
)
|
||||
)
|
||||
if block_id in self.out_indices:
|
||||
self.out_channels.append(out_channels)
|
||||
|
||||
if not self.det:
|
||||
self.out_channels = stage_config["stage4"][2]
|
||||
|
||||
self._init_weights()
|
||||
|
||||
def _init_weights(self):
|
||||
for m in self.sublayers():
|
||||
if isinstance(m, nn.Conv2D):
|
||||
kaiming_normal_(m.weight)
|
||||
elif isinstance(m, (nn.BatchNorm2D)):
|
||||
ones_(m.weight)
|
||||
zeros_(m.bias)
|
||||
elif isinstance(m, nn.Linear):
|
||||
zeros_(m.bias)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.stem(x)
|
||||
if self.det:
|
||||
x = self.pool(x)
|
||||
|
||||
out = []
|
||||
for i, stage in enumerate(self.stages):
|
||||
x = stage(x)
|
||||
if self.det and i in self.out_indices:
|
||||
out.append(x)
|
||||
if self.det:
|
||||
return out
|
||||
|
||||
if self.training:
|
||||
x = F.adaptive_avg_pool2d(x, [1, 40])
|
||||
else:
|
||||
x = F.avg_pool2d(x, [3, 2])
|
||||
return x
|
||||
|
||||
|
||||
def PPHGNet_tiny(pretrained=False, use_ssld=False, **kwargs):
|
||||
"""
|
||||
PPHGNet_tiny
|
||||
Args:
|
||||
pretrained: bool=False or str. If `True` load pretrained parameters, `False` otherwise.
|
||||
If str, means the path of the pretrained model.
|
||||
use_ssld: bool=False. Whether using distillation pretrained model when pretrained=True.
|
||||
Returns:
|
||||
model: nn.Layer. Specific `PPHGNet_tiny` model depends on args.
|
||||
"""
|
||||
stage_config = {
|
||||
# in_channels, mid_channels, out_channels, blocks, downsample
|
||||
"stage1": [96, 96, 224, 1, False, [2, 1]],
|
||||
"stage2": [224, 128, 448, 1, True, [1, 2]],
|
||||
"stage3": [448, 160, 512, 2, True, [2, 1]],
|
||||
"stage4": [512, 192, 768, 1, True, [2, 1]],
|
||||
}
|
||||
|
||||
model = PPHGNet(
|
||||
stem_channels=[48, 48, 96], stage_config=stage_config, layer_num=5, **kwargs
|
||||
)
|
||||
return model
|
||||
|
||||
|
||||
def PPHGNet_small(pretrained=False, use_ssld=False, det=False, **kwargs):
|
||||
"""
|
||||
PPHGNet_small
|
||||
Args:
|
||||
pretrained: bool=False or str. If `True` load pretrained parameters, `False` otherwise.
|
||||
If str, means the path of the pretrained model.
|
||||
use_ssld: bool=False. Whether using distillation pretrained model when pretrained=True.
|
||||
Returns:
|
||||
model: nn.Layer. Specific `PPHGNet_small` model depends on args.
|
||||
"""
|
||||
stage_config_det = {
|
||||
# in_channels, mid_channels, out_channels, blocks, downsample
|
||||
"stage1": [128, 128, 256, 1, False, 2],
|
||||
"stage2": [256, 160, 512, 1, True, 2],
|
||||
"stage3": [512, 192, 768, 2, True, 2],
|
||||
"stage4": [768, 224, 1024, 1, True, 2],
|
||||
}
|
||||
|
||||
stage_config_rec = {
|
||||
# in_channels, mid_channels, out_channels, blocks, downsample
|
||||
"stage1": [128, 128, 256, 1, True, [2, 1]],
|
||||
"stage2": [256, 160, 512, 1, True, [1, 2]],
|
||||
"stage3": [512, 192, 768, 2, True, [2, 1]],
|
||||
"stage4": [768, 224, 1024, 1, True, [2, 1]],
|
||||
}
|
||||
|
||||
model = PPHGNet(
|
||||
stem_channels=[64, 64, 128],
|
||||
stage_config=stage_config_det if det else stage_config_rec,
|
||||
layer_num=6,
|
||||
det=det,
|
||||
**kwargs,
|
||||
)
|
||||
return model
|
||||
|
||||
|
||||
def PPHGNet_base(pretrained=False, use_ssld=True, **kwargs):
|
||||
"""
|
||||
PPHGNet_base
|
||||
Args:
|
||||
pretrained: bool=False or str. If `True` load pretrained parameters, `False` otherwise.
|
||||
If str, means the path of the pretrained model.
|
||||
use_ssld: bool=False. Whether using distillation pretrained model when pretrained=True.
|
||||
Returns:
|
||||
model: nn.Layer. Specific `PPHGNet_base` model depends on args.
|
||||
"""
|
||||
stage_config = {
|
||||
# in_channels, mid_channels, out_channels, blocks, downsample
|
||||
"stage1": [160, 192, 320, 1, False, [2, 1]],
|
||||
"stage2": [320, 224, 640, 2, True, [1, 2]],
|
||||
"stage3": [640, 256, 960, 3, True, [2, 1]],
|
||||
"stage4": [960, 288, 1280, 2, True, [2, 1]],
|
||||
}
|
||||
|
||||
model = PPHGNet(
|
||||
stem_channels=[96, 96, 160],
|
||||
stage_config=stage_config,
|
||||
layer_num=7,
|
||||
dropout_prob=0.2,
|
||||
**kwargs,
|
||||
)
|
||||
return model
|
||||
529
ppocr/modeling/backbones/rec_hybridvit.py
Normal file
529
ppocr/modeling/backbones/rec_hybridvit.py
Normal file
@@ -0,0 +1,529 @@
|
||||
# copyright (c) 2024 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/vision_transformer_hybrid.py
|
||||
"""
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
from itertools import repeat
|
||||
import collections
|
||||
import math
|
||||
from functools import partial
|
||||
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
from ppocr.modeling.backbones.rec_resnetv2 import (
|
||||
ResNetV2,
|
||||
StdConv2dSame,
|
||||
DropPath,
|
||||
get_padding,
|
||||
)
|
||||
from paddle.nn.initializer import (
|
||||
TruncatedNormal,
|
||||
Constant,
|
||||
Normal,
|
||||
KaimingUniform,
|
||||
XavierUniform,
|
||||
)
|
||||
|
||||
normal_ = Normal(mean=0.0, std=1e-6)
|
||||
zeros_ = Constant(value=0.0)
|
||||
ones_ = Constant(value=1.0)
|
||||
kaiming_normal_ = KaimingUniform(nonlinearity="relu")
|
||||
trunc_normal_ = TruncatedNormal(std=0.02)
|
||||
xavier_uniform_ = XavierUniform()
|
||||
|
||||
|
||||
def _ntuple(n):
|
||||
def parse(x):
|
||||
if isinstance(x, collections.abc.Iterable):
|
||||
return x
|
||||
return tuple(repeat(x, n))
|
||||
|
||||
return parse
|
||||
|
||||
|
||||
to_1tuple = _ntuple(1)
|
||||
to_2tuple = _ntuple(2)
|
||||
to_3tuple = _ntuple(3)
|
||||
to_4tuple = _ntuple(4)
|
||||
to_ntuple = _ntuple
|
||||
|
||||
|
||||
class Conv2dAlign(nn.Conv2D):
|
||||
"""Conv2d with Weight Standardization. Used for BiT ResNet-V2 models.
|
||||
|
||||
Paper: `Micro-Batch Training with Batch-Channel Normalization and Weight Standardization` -
|
||||
https://arxiv.org/abs/1903.10520v2
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
in_channel,
|
||||
out_channels,
|
||||
kernel_size,
|
||||
stride=1,
|
||||
padding=0,
|
||||
dilation=1,
|
||||
groups=1,
|
||||
bias=True,
|
||||
eps=1e-6,
|
||||
):
|
||||
|
||||
super().__init__(
|
||||
in_channel,
|
||||
out_channels,
|
||||
kernel_size,
|
||||
stride=stride,
|
||||
padding=padding,
|
||||
dilation=dilation,
|
||||
groups=groups,
|
||||
bias_attr=bias,
|
||||
weight_attr=True,
|
||||
)
|
||||
self.eps = eps
|
||||
|
||||
def forward(self, x):
|
||||
x = F.conv2d(
|
||||
x,
|
||||
self.weight,
|
||||
self.bias,
|
||||
self._stride,
|
||||
self._padding,
|
||||
self._dilation,
|
||||
self._groups,
|
||||
)
|
||||
return x
|
||||
|
||||
|
||||
class HybridEmbed(nn.Layer):
|
||||
"""CNN Feature Map Embedding
|
||||
Extract feature map from CNN, flatten, project to embedding dim.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
backbone,
|
||||
img_size=224,
|
||||
patch_size=1,
|
||||
feature_size=None,
|
||||
in_chans=3,
|
||||
embed_dim=768,
|
||||
):
|
||||
super().__init__()
|
||||
assert isinstance(backbone, nn.Layer)
|
||||
img_size = to_2tuple(img_size)
|
||||
patch_size = to_2tuple(patch_size)
|
||||
self.img_size = img_size
|
||||
self.patch_size = patch_size
|
||||
self.backbone = backbone
|
||||
feature_dim = 1024
|
||||
feature_size = (42, 12)
|
||||
patch_size = (1, 1)
|
||||
assert (
|
||||
feature_size[0] % patch_size[0] == 0
|
||||
and feature_size[1] % patch_size[1] == 0
|
||||
)
|
||||
|
||||
self.grid_size = (
|
||||
feature_size[0] // patch_size[0],
|
||||
feature_size[1] // patch_size[1],
|
||||
)
|
||||
self.num_patches = self.grid_size[0] * self.grid_size[1]
|
||||
self.proj = nn.Conv2D(
|
||||
feature_dim,
|
||||
embed_dim,
|
||||
kernel_size=patch_size,
|
||||
stride=patch_size,
|
||||
weight_attr=True,
|
||||
bias_attr=True,
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
|
||||
x = self.backbone(x)
|
||||
if isinstance(x, (list, tuple)):
|
||||
x = x[-1] # last feature if backbone outputs list/tuple of features
|
||||
x = self.proj(x).flatten(2).transpose([0, 2, 1])
|
||||
|
||||
return x
|
||||
|
||||
|
||||
class myLinear(nn.Linear):
|
||||
def __init__(self, in_channel, out_channels, weight_attr=True, bias_attr=True):
|
||||
super().__init__(
|
||||
in_channel, out_channels, weight_attr=weight_attr, bias_attr=bias_attr
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
return paddle.matmul(x, self.weight, transpose_y=True) + self.bias
|
||||
|
||||
|
||||
class Attention(nn.Layer):
|
||||
def __init__(self, dim, num_heads=8, qkv_bias=False, attn_drop=0.0, proj_drop=0.0):
|
||||
super().__init__()
|
||||
self.num_heads = num_heads
|
||||
head_dim = dim // num_heads
|
||||
self.scale = head_dim**-0.5
|
||||
|
||||
self.qkv = nn.Linear(dim, dim * 3, bias_attr=qkv_bias)
|
||||
self.attn_drop = nn.Dropout(attn_drop)
|
||||
self.proj = myLinear(dim, dim, weight_attr=True, bias_attr=True)
|
||||
self.proj_drop = nn.Dropout(proj_drop)
|
||||
|
||||
def forward(self, x):
|
||||
B, N, C = x.shape
|
||||
qkv = (
|
||||
self.qkv(x)
|
||||
.reshape([B, N, 3, self.num_heads, C // self.num_heads])
|
||||
.transpose([2, 0, 3, 1, 4])
|
||||
)
|
||||
q, k, v = qkv.unbind(0) # make torchscript happy (cannot use tensor as tuple)
|
||||
|
||||
attn = (q @ k.transpose([0, 1, 3, 2])) * self.scale
|
||||
|
||||
attn = F.softmax(attn, axis=-1)
|
||||
attn = self.attn_drop(attn)
|
||||
|
||||
x = (attn @ v).transpose([0, 2, 1, 3]).reshape([B, N, C])
|
||||
|
||||
x = self.proj(x)
|
||||
x = self.proj_drop(x)
|
||||
return x
|
||||
|
||||
|
||||
class Mlp(nn.Layer):
|
||||
"""MLP as used in Vision Transformer, MLP-Mixer and related networks"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
in_features,
|
||||
hidden_features=None,
|
||||
out_features=None,
|
||||
act_layer=nn.GELU,
|
||||
drop=0.0,
|
||||
):
|
||||
super().__init__()
|
||||
out_features = out_features or in_features
|
||||
hidden_features = hidden_features or in_features
|
||||
drop_probs = to_2tuple(drop)
|
||||
|
||||
self.fc1 = nn.Linear(in_features, hidden_features)
|
||||
self.act = act_layer()
|
||||
self.drop1 = nn.Dropout(drop_probs[0])
|
||||
self.fc2 = nn.Linear(hidden_features, out_features)
|
||||
self.drop2 = nn.Dropout(drop_probs[1])
|
||||
|
||||
def forward(self, x):
|
||||
x = self.fc1(x)
|
||||
x = self.act(x)
|
||||
x = self.drop1(x)
|
||||
x = self.fc2(x)
|
||||
x = self.drop2(x)
|
||||
return x
|
||||
|
||||
|
||||
class Block(nn.Layer):
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
dim,
|
||||
num_heads,
|
||||
mlp_ratio=4.0,
|
||||
qkv_bias=False,
|
||||
drop=0.0,
|
||||
attn_drop=0.0,
|
||||
drop_path=0.0,
|
||||
act_layer=nn.GELU,
|
||||
norm_layer=nn.LayerNorm,
|
||||
):
|
||||
super().__init__()
|
||||
self.norm1 = norm_layer(dim)
|
||||
self.attn = Attention(
|
||||
dim,
|
||||
num_heads=num_heads,
|
||||
qkv_bias=qkv_bias,
|
||||
attn_drop=attn_drop,
|
||||
proj_drop=drop,
|
||||
)
|
||||
# NOTE: drop path for stochastic depth, we shall see if this is better than dropout here
|
||||
self.drop_path = DropPath(drop_path) if drop_path > 0.0 else nn.Identity()
|
||||
self.norm2 = norm_layer(dim)
|
||||
mlp_hidden_dim = int(dim * mlp_ratio)
|
||||
self.mlp = Mlp(
|
||||
in_features=dim,
|
||||
hidden_features=mlp_hidden_dim,
|
||||
act_layer=act_layer,
|
||||
drop=drop,
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
|
||||
x = x + self.drop_path(self.attn(self.norm1(x)))
|
||||
x = x + self.drop_path(self.mlp(self.norm2(x)))
|
||||
return x
|
||||
|
||||
|
||||
class HybridTransformer(nn.Layer):
|
||||
"""Implementation of HybridTransformer.
|
||||
|
||||
Args:
|
||||
x: input images with shape [N, 1, H, W]
|
||||
label: LaTeX-OCR labels with shape [N, L] , L is the max sequence length
|
||||
attention_mask: LaTeX-OCR attention mask with shape [N, L] , L is the max sequence length
|
||||
|
||||
Returns:
|
||||
The encoded features with shape [N, 1, H//16, W//16]
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
backbone_layers=[2, 3, 7],
|
||||
input_channel=1,
|
||||
is_predict=False,
|
||||
is_export=False,
|
||||
img_size=(224, 224),
|
||||
patch_size=16,
|
||||
num_classes=1000,
|
||||
embed_dim=768,
|
||||
depth=12,
|
||||
num_heads=12,
|
||||
mlp_ratio=4.0,
|
||||
qkv_bias=True,
|
||||
representation_size=None,
|
||||
distilled=False,
|
||||
drop_rate=0.0,
|
||||
attn_drop_rate=0.0,
|
||||
drop_path_rate=0.0,
|
||||
embed_layer=None,
|
||||
norm_layer=None,
|
||||
act_layer=None,
|
||||
weight_init="",
|
||||
**kwargs,
|
||||
):
|
||||
super(HybridTransformer, self).__init__()
|
||||
self.num_classes = num_classes
|
||||
self.num_features = self.embed_dim = (
|
||||
embed_dim # num_features for consistency with other models
|
||||
)
|
||||
self.num_tokens = 2 if distilled else 1
|
||||
norm_layer = norm_layer or partial(nn.LayerNorm, epsilon=1e-6)
|
||||
act_layer = act_layer or nn.GELU
|
||||
self.height, self.width = img_size
|
||||
self.patch_size = patch_size
|
||||
backbone = ResNetV2(
|
||||
layers=backbone_layers,
|
||||
num_classes=0,
|
||||
global_pool="",
|
||||
in_chans=input_channel,
|
||||
preact=False,
|
||||
stem_type="same",
|
||||
conv_layer=StdConv2dSame,
|
||||
is_export=is_export,
|
||||
)
|
||||
min_patch_size = 2 ** (len(backbone_layers) + 1)
|
||||
self.patch_embed = HybridEmbed(
|
||||
img_size=img_size,
|
||||
patch_size=patch_size // min_patch_size,
|
||||
in_chans=input_channel,
|
||||
embed_dim=embed_dim,
|
||||
backbone=backbone,
|
||||
)
|
||||
num_patches = self.patch_embed.num_patches
|
||||
|
||||
self.cls_token = paddle.create_parameter([1, 1, embed_dim], dtype="float32")
|
||||
self.dist_token = (
|
||||
paddle.create_parameter(
|
||||
[1, 1, embed_dim],
|
||||
dtype="float32",
|
||||
)
|
||||
if distilled
|
||||
else None
|
||||
)
|
||||
self.pos_embed = paddle.create_parameter(
|
||||
[1, num_patches + self.num_tokens, embed_dim], dtype="float32"
|
||||
)
|
||||
self.pos_drop = nn.Dropout(p=drop_rate)
|
||||
zeros_(self.cls_token)
|
||||
if self.dist_token is not None:
|
||||
zeros_(self.dist_token)
|
||||
zeros_(self.pos_embed)
|
||||
|
||||
dpr = [
|
||||
x.item() for x in paddle.linspace(0, drop_path_rate, depth)
|
||||
] # stochastic depth decay rule
|
||||
self.blocks = nn.Sequential(
|
||||
*[
|
||||
Block(
|
||||
dim=embed_dim,
|
||||
num_heads=num_heads,
|
||||
mlp_ratio=mlp_ratio,
|
||||
qkv_bias=qkv_bias,
|
||||
drop=drop_rate,
|
||||
attn_drop=attn_drop_rate,
|
||||
drop_path=dpr[i],
|
||||
norm_layer=norm_layer,
|
||||
act_layer=act_layer,
|
||||
)
|
||||
for i in range(depth)
|
||||
]
|
||||
)
|
||||
self.norm = norm_layer(embed_dim)
|
||||
|
||||
# Representation layer
|
||||
if representation_size and not distilled:
|
||||
self.num_features = representation_size
|
||||
self.pre_logits = nn.Sequential(
|
||||
("fc", nn.Linear(embed_dim, representation_size)), ("act", nn.Tanh())
|
||||
)
|
||||
else:
|
||||
self.pre_logits = nn.Identity()
|
||||
|
||||
# Classifier head(s)
|
||||
self.head = (
|
||||
nn.Linear(self.num_features, num_classes)
|
||||
if num_classes > 0
|
||||
else nn.Identity()
|
||||
)
|
||||
self.head_dist = None
|
||||
if distilled:
|
||||
self.head_dist = (
|
||||
nn.Linear(self.embed_dim, self.num_classes)
|
||||
if num_classes > 0
|
||||
else nn.Identity()
|
||||
)
|
||||
self.init_weights(weight_init)
|
||||
self.out_channels = embed_dim
|
||||
self.is_predict = is_predict
|
||||
self.is_export = is_export
|
||||
|
||||
def init_weights(self, mode=""):
|
||||
assert mode in ("jax", "jax_nlhb", "nlhb", "")
|
||||
head_bias = -math.log(self.num_classes) if "nlhb" in mode else 0.0
|
||||
trunc_normal_(self.pos_embed)
|
||||
trunc_normal_(self.cls_token)
|
||||
self.apply(_init_vit_weights)
|
||||
|
||||
def _init_weights(self, m):
|
||||
# this fn left here for compat with downstream users
|
||||
_init_vit_weights(m)
|
||||
|
||||
def load_pretrained(self, checkpoint_path, prefix=""):
|
||||
raise NotImplementedError
|
||||
|
||||
def no_weight_decay(self):
|
||||
return {"pos_embed", "cls_token", "dist_token"}
|
||||
|
||||
def get_classifier(self):
|
||||
if self.dist_token is None:
|
||||
return self.head
|
||||
else:
|
||||
return self.head, self.head_dist
|
||||
|
||||
def reset_classifier(self, num_classes, global_pool=""):
|
||||
self.num_classes = num_classes
|
||||
self.head = (
|
||||
nn.Linear(self.embed_dim, num_classes) if num_classes > 0 else nn.Identity()
|
||||
)
|
||||
if self.num_tokens == 2:
|
||||
self.head_dist = (
|
||||
nn.Linear(self.embed_dim, self.num_classes)
|
||||
if num_classes > 0
|
||||
else nn.Identity()
|
||||
)
|
||||
|
||||
def forward_features(self, x):
|
||||
B, c, h, w = x.shape
|
||||
x = self.patch_embed(x)
|
||||
cls_tokens = self.cls_token.expand(
|
||||
[B, -1, -1]
|
||||
) # stole cls_tokens impl from Phil Wang, thanks
|
||||
x = paddle.concat((cls_tokens, x), axis=1)
|
||||
h, w = h // self.patch_size, w // self.patch_size
|
||||
repeat_tensor = (
|
||||
paddle.arange(h) * (self.width // self.patch_size - w)
|
||||
).reshape([-1, 1])
|
||||
repeat_tensor = paddle.repeat_interleave(
|
||||
repeat_tensor, paddle.to_tensor(w), axis=1
|
||||
).reshape([-1])
|
||||
pos_emb_ind = repeat_tensor + paddle.arange(h * w)
|
||||
pos_emb_ind = paddle.concat(
|
||||
(paddle.zeros([1], dtype="int64"), pos_emb_ind + 1), axis=0
|
||||
).cast(paddle.int64)
|
||||
x += self.pos_embed[:, pos_emb_ind]
|
||||
x = self.pos_drop(x)
|
||||
|
||||
for blk in self.blocks:
|
||||
x = blk(x)
|
||||
|
||||
x = self.norm(x)
|
||||
return x
|
||||
|
||||
def forward(self, input_data):
|
||||
|
||||
if self.training:
|
||||
x, label, attention_mask = input_data
|
||||
else:
|
||||
if isinstance(input_data, list):
|
||||
x = input_data[0]
|
||||
else:
|
||||
x = input_data
|
||||
x = self.forward_features(x)
|
||||
x = self.head(x)
|
||||
if self.training:
|
||||
return x, label, attention_mask
|
||||
else:
|
||||
return x
|
||||
|
||||
|
||||
def _init_vit_weights(
|
||||
module: nn.Layer, name: str = "", head_bias: float = 0.0, jax_impl: bool = False
|
||||
):
|
||||
"""ViT weight initialization
|
||||
* When called without n, head_bias, jax_impl args it will behave exactly the same
|
||||
as my original init for compatibility with prev hparam / downstream use cases (ie DeiT).
|
||||
* When called w/ valid n (module name) and jax_impl=True, will (hopefully) match JAX impl
|
||||
"""
|
||||
if isinstance(module, nn.Linear):
|
||||
if name.startswith("head"):
|
||||
zeros_(module.weight)
|
||||
constant_ = Constant(value=head_bias)
|
||||
constant_(module.bias, head_bias)
|
||||
elif name.startswith("pre_logits"):
|
||||
zeros_(module.bias)
|
||||
else:
|
||||
if jax_impl:
|
||||
xavier_uniform_(module.weight)
|
||||
if module.bias is not None:
|
||||
if "mlp" in name:
|
||||
normal_(module.bias)
|
||||
else:
|
||||
zeros_(module.bias)
|
||||
else:
|
||||
trunc_normal_(module.weight)
|
||||
if module.bias is not None:
|
||||
zeros_(module.bias)
|
||||
elif jax_impl and isinstance(module, nn.Conv2D):
|
||||
# NOTE conv was left to pytorch default in my original init
|
||||
if module.bias is not None:
|
||||
zeros_(module.bias)
|
||||
elif isinstance(module, (nn.LayerNorm, nn.GroupNorm, nn.BatchNorm2D)):
|
||||
zeros_(module.bias)
|
||||
ones_(module.weight)
|
||||
558
ppocr/modeling/backbones/rec_lcnetv3.py
Normal file
558
ppocr/modeling/backbones/rec_lcnetv3.py
Normal file
@@ -0,0 +1,558 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
from paddle import ParamAttr
|
||||
from paddle.nn.initializer import Constant, KaimingNormal
|
||||
from paddle.nn import (
|
||||
AdaptiveAvgPool2D,
|
||||
BatchNorm2D,
|
||||
Conv2D,
|
||||
Dropout,
|
||||
Hardsigmoid,
|
||||
Hardswish,
|
||||
Identity,
|
||||
Linear,
|
||||
ReLU,
|
||||
)
|
||||
from paddle.regularizer import L2Decay
|
||||
from ppocr.modeling.backbones.rec_hgnet import MeanPool2D
|
||||
|
||||
NET_CONFIG_det = {
|
||||
"blocks2":
|
||||
# k, in_c, out_c, s, use_se
|
||||
[[3, 16, 32, 1, False]],
|
||||
"blocks3": [[3, 32, 64, 2, False], [3, 64, 64, 1, False]],
|
||||
"blocks4": [[3, 64, 128, 2, False], [3, 128, 128, 1, False]],
|
||||
"blocks5": [
|
||||
[3, 128, 256, 2, False],
|
||||
[5, 256, 256, 1, False],
|
||||
[5, 256, 256, 1, False],
|
||||
[5, 256, 256, 1, False],
|
||||
[5, 256, 256, 1, False],
|
||||
],
|
||||
"blocks6": [
|
||||
[5, 256, 512, 2, True],
|
||||
[5, 512, 512, 1, True],
|
||||
[5, 512, 512, 1, False],
|
||||
[5, 512, 512, 1, False],
|
||||
],
|
||||
}
|
||||
|
||||
NET_CONFIG_rec = {
|
||||
"blocks2":
|
||||
# k, in_c, out_c, s, use_se
|
||||
[[3, 16, 32, 1, False]],
|
||||
"blocks3": [[3, 32, 64, 1, False], [3, 64, 64, 1, False]],
|
||||
"blocks4": [[3, 64, 128, (2, 1), False], [3, 128, 128, 1, False]],
|
||||
"blocks5": [
|
||||
[3, 128, 256, (1, 2), False],
|
||||
[5, 256, 256, 1, False],
|
||||
[5, 256, 256, 1, False],
|
||||
[5, 256, 256, 1, False],
|
||||
[5, 256, 256, 1, False],
|
||||
],
|
||||
"blocks6": [
|
||||
[5, 256, 512, (2, 1), True],
|
||||
[5, 512, 512, 1, True],
|
||||
[5, 512, 512, (2, 1), False],
|
||||
[5, 512, 512, 1, False],
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def make_divisible(v, divisor=16, min_value=None):
|
||||
if min_value is None:
|
||||
min_value = divisor
|
||||
new_v = max(min_value, int(v + divisor / 2) // divisor * divisor)
|
||||
if new_v < 0.9 * v:
|
||||
new_v += divisor
|
||||
return new_v
|
||||
|
||||
|
||||
class LearnableAffineBlock(nn.Layer):
|
||||
def __init__(self, scale_value=1.0, bias_value=0.0, lr_mult=1.0, lab_lr=0.1):
|
||||
super().__init__()
|
||||
self.scale = self.create_parameter(
|
||||
shape=[
|
||||
1,
|
||||
],
|
||||
default_initializer=Constant(value=scale_value),
|
||||
attr=ParamAttr(learning_rate=lr_mult * lab_lr),
|
||||
)
|
||||
self.add_parameter("scale", self.scale)
|
||||
self.bias = self.create_parameter(
|
||||
shape=[
|
||||
1,
|
||||
],
|
||||
default_initializer=Constant(value=bias_value),
|
||||
attr=ParamAttr(learning_rate=lr_mult * lab_lr),
|
||||
)
|
||||
self.add_parameter("bias", self.bias)
|
||||
|
||||
def forward(self, x):
|
||||
return self.scale * x + self.bias
|
||||
|
||||
|
||||
class ConvBNLayer(nn.Layer):
|
||||
def __init__(
|
||||
self, in_channels, out_channels, kernel_size, stride, groups=1, lr_mult=1.0
|
||||
):
|
||||
super().__init__()
|
||||
self.conv = Conv2D(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=kernel_size,
|
||||
stride=stride,
|
||||
padding=(kernel_size - 1) // 2,
|
||||
groups=groups,
|
||||
weight_attr=ParamAttr(initializer=KaimingNormal(), learning_rate=lr_mult),
|
||||
bias_attr=False,
|
||||
)
|
||||
|
||||
self.bn = BatchNorm2D(
|
||||
out_channels,
|
||||
weight_attr=ParamAttr(regularizer=L2Decay(0.0), learning_rate=lr_mult),
|
||||
bias_attr=ParamAttr(regularizer=L2Decay(0.0), learning_rate=lr_mult),
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.conv(x)
|
||||
x = self.bn(x)
|
||||
return x
|
||||
|
||||
|
||||
class Act(nn.Layer):
|
||||
def __init__(self, act="hswish", lr_mult=1.0, lab_lr=0.1):
|
||||
super().__init__()
|
||||
if act == "hswish":
|
||||
self.act = Hardswish()
|
||||
else:
|
||||
assert act == "relu"
|
||||
self.act = ReLU()
|
||||
self.lab = LearnableAffineBlock(lr_mult=lr_mult, lab_lr=lab_lr)
|
||||
|
||||
def forward(self, x):
|
||||
return self.lab(self.act(x))
|
||||
|
||||
|
||||
class LearnableRepLayer(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
kernel_size,
|
||||
stride=1,
|
||||
groups=1,
|
||||
num_conv_branches=1,
|
||||
lr_mult=1.0,
|
||||
lab_lr=0.1,
|
||||
):
|
||||
super().__init__()
|
||||
self.is_repped = False
|
||||
self.groups = groups
|
||||
self.stride = stride
|
||||
self.kernel_size = kernel_size
|
||||
self.in_channels = in_channels
|
||||
self.out_channels = out_channels
|
||||
self.num_conv_branches = num_conv_branches
|
||||
self.padding = (kernel_size - 1) // 2
|
||||
|
||||
self.identity = (
|
||||
BatchNorm2D(
|
||||
num_features=in_channels,
|
||||
weight_attr=ParamAttr(learning_rate=lr_mult),
|
||||
bias_attr=ParamAttr(learning_rate=lr_mult),
|
||||
)
|
||||
if out_channels == in_channels and stride == 1
|
||||
else None
|
||||
)
|
||||
|
||||
self.conv_kxk = nn.LayerList(
|
||||
[
|
||||
ConvBNLayer(
|
||||
in_channels,
|
||||
out_channels,
|
||||
kernel_size,
|
||||
stride,
|
||||
groups=groups,
|
||||
lr_mult=lr_mult,
|
||||
)
|
||||
for _ in range(self.num_conv_branches)
|
||||
]
|
||||
)
|
||||
|
||||
self.conv_1x1 = (
|
||||
ConvBNLayer(
|
||||
in_channels, out_channels, 1, stride, groups=groups, lr_mult=lr_mult
|
||||
)
|
||||
if kernel_size > 1
|
||||
else None
|
||||
)
|
||||
|
||||
self.lab = LearnableAffineBlock(lr_mult=lr_mult, lab_lr=lab_lr)
|
||||
self.act = Act(lr_mult=lr_mult, lab_lr=lab_lr)
|
||||
|
||||
def forward(self, x):
|
||||
# for export
|
||||
if self.is_repped:
|
||||
out = self.lab(self.reparam_conv(x))
|
||||
if self.stride != 2:
|
||||
out = self.act(out)
|
||||
return out
|
||||
|
||||
out = 0
|
||||
if self.identity is not None:
|
||||
out += self.identity(x)
|
||||
|
||||
if self.conv_1x1 is not None:
|
||||
out += self.conv_1x1(x)
|
||||
|
||||
for conv in self.conv_kxk:
|
||||
out += conv(x)
|
||||
|
||||
out = self.lab(out)
|
||||
if self.stride != 2:
|
||||
out = self.act(out)
|
||||
return out
|
||||
|
||||
def rep(self):
|
||||
if self.is_repped:
|
||||
return
|
||||
kernel, bias = self._get_kernel_bias()
|
||||
self.reparam_conv = Conv2D(
|
||||
in_channels=self.in_channels,
|
||||
out_channels=self.out_channels,
|
||||
kernel_size=self.kernel_size,
|
||||
stride=self.stride,
|
||||
padding=self.padding,
|
||||
groups=self.groups,
|
||||
)
|
||||
self.reparam_conv.weight.set_value(kernel)
|
||||
self.reparam_conv.bias.set_value(bias)
|
||||
self.is_repped = True
|
||||
|
||||
def _pad_kernel_1x1_to_kxk(self, kernel1x1, pad):
|
||||
if not isinstance(kernel1x1, paddle.Tensor):
|
||||
return 0
|
||||
else:
|
||||
return nn.functional.pad(kernel1x1, [pad, pad, pad, pad])
|
||||
|
||||
def _get_kernel_bias(self):
|
||||
kernel_conv_1x1, bias_conv_1x1 = self._fuse_bn_tensor(self.conv_1x1)
|
||||
kernel_conv_1x1 = self._pad_kernel_1x1_to_kxk(
|
||||
kernel_conv_1x1, self.kernel_size // 2
|
||||
)
|
||||
|
||||
kernel_identity, bias_identity = self._fuse_bn_tensor(self.identity)
|
||||
|
||||
kernel_conv_kxk = 0
|
||||
bias_conv_kxk = 0
|
||||
for conv in self.conv_kxk:
|
||||
kernel, bias = self._fuse_bn_tensor(conv)
|
||||
kernel_conv_kxk += kernel
|
||||
bias_conv_kxk += bias
|
||||
|
||||
kernel_reparam = kernel_conv_kxk + kernel_conv_1x1 + kernel_identity
|
||||
bias_reparam = bias_conv_kxk + bias_conv_1x1 + bias_identity
|
||||
return kernel_reparam, bias_reparam
|
||||
|
||||
def _fuse_bn_tensor(self, branch):
|
||||
if not branch:
|
||||
return 0, 0
|
||||
elif isinstance(branch, ConvBNLayer):
|
||||
kernel = branch.conv.weight
|
||||
running_mean = branch.bn._mean
|
||||
running_var = branch.bn._variance
|
||||
gamma = branch.bn.weight
|
||||
beta = branch.bn.bias
|
||||
eps = branch.bn._epsilon
|
||||
else:
|
||||
assert isinstance(branch, BatchNorm2D)
|
||||
if not hasattr(self, "id_tensor"):
|
||||
input_dim = self.in_channels // self.groups
|
||||
kernel_value = paddle.zeros(
|
||||
(self.in_channels, input_dim, self.kernel_size, self.kernel_size),
|
||||
dtype=branch.weight.dtype,
|
||||
)
|
||||
for i in range(self.in_channels):
|
||||
kernel_value[
|
||||
i, i % input_dim, self.kernel_size // 2, self.kernel_size // 2
|
||||
] = 1
|
||||
self.id_tensor = kernel_value
|
||||
kernel = self.id_tensor
|
||||
running_mean = branch._mean
|
||||
running_var = branch._variance
|
||||
gamma = branch.weight
|
||||
beta = branch.bias
|
||||
eps = branch._epsilon
|
||||
std = (running_var + eps).sqrt()
|
||||
t = (gamma / std).reshape((-1, 1, 1, 1))
|
||||
return kernel * t, beta - running_mean * gamma / std
|
||||
|
||||
|
||||
class SELayer(nn.Layer):
|
||||
def __init__(self, channel, reduction=4, lr_mult=1.0):
|
||||
super().__init__()
|
||||
if "npu" in paddle.device.get_device():
|
||||
self.avg_pool = MeanPool2D(1, 1)
|
||||
else:
|
||||
self.avg_pool = AdaptiveAvgPool2D(1)
|
||||
self.conv1 = Conv2D(
|
||||
in_channels=channel,
|
||||
out_channels=channel // reduction,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
padding=0,
|
||||
weight_attr=ParamAttr(learning_rate=lr_mult),
|
||||
bias_attr=ParamAttr(learning_rate=lr_mult),
|
||||
)
|
||||
self.relu = ReLU()
|
||||
self.conv2 = Conv2D(
|
||||
in_channels=channel // reduction,
|
||||
out_channels=channel,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
padding=0,
|
||||
weight_attr=ParamAttr(learning_rate=lr_mult),
|
||||
bias_attr=ParamAttr(learning_rate=lr_mult),
|
||||
)
|
||||
self.hardsigmoid = Hardsigmoid()
|
||||
|
||||
def forward(self, x):
|
||||
identity = x
|
||||
x = self.avg_pool(x)
|
||||
x = self.conv1(x)
|
||||
x = self.relu(x)
|
||||
x = self.conv2(x)
|
||||
x = self.hardsigmoid(x)
|
||||
x = paddle.multiply(x=identity, y=x)
|
||||
return x
|
||||
|
||||
|
||||
class LCNetV3Block(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
stride,
|
||||
dw_size,
|
||||
use_se=False,
|
||||
conv_kxk_num=4,
|
||||
lr_mult=1.0,
|
||||
lab_lr=0.1,
|
||||
):
|
||||
super().__init__()
|
||||
self.use_se = use_se
|
||||
self.dw_conv = LearnableRepLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=in_channels,
|
||||
kernel_size=dw_size,
|
||||
stride=stride,
|
||||
groups=in_channels,
|
||||
num_conv_branches=conv_kxk_num,
|
||||
lr_mult=lr_mult,
|
||||
lab_lr=lab_lr,
|
||||
)
|
||||
if use_se:
|
||||
self.se = SELayer(in_channels, lr_mult=lr_mult)
|
||||
self.pw_conv = LearnableRepLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
num_conv_branches=conv_kxk_num,
|
||||
lr_mult=lr_mult,
|
||||
lab_lr=lab_lr,
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.dw_conv(x)
|
||||
if self.use_se:
|
||||
x = self.se(x)
|
||||
x = self.pw_conv(x)
|
||||
return x
|
||||
|
||||
|
||||
class PPLCNetV3(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
scale=1.0,
|
||||
conv_kxk_num=4,
|
||||
lr_mult_list=[1.0, 1.0, 1.0, 1.0, 1.0, 1.0],
|
||||
lab_lr=0.1,
|
||||
det=False,
|
||||
**kwargs,
|
||||
):
|
||||
super().__init__()
|
||||
self.scale = scale
|
||||
self.lr_mult_list = lr_mult_list
|
||||
self.det = det
|
||||
|
||||
self.net_config = NET_CONFIG_det if self.det else NET_CONFIG_rec
|
||||
|
||||
assert isinstance(
|
||||
self.lr_mult_list, (list, tuple)
|
||||
), "lr_mult_list should be in (list, tuple) but got {}".format(
|
||||
type(self.lr_mult_list)
|
||||
)
|
||||
assert (
|
||||
len(self.lr_mult_list) == 6
|
||||
), "lr_mult_list length should be 6 but got {}".format(len(self.lr_mult_list))
|
||||
|
||||
self.conv1 = ConvBNLayer(
|
||||
in_channels=3,
|
||||
out_channels=make_divisible(16 * scale),
|
||||
kernel_size=3,
|
||||
stride=2,
|
||||
lr_mult=self.lr_mult_list[0],
|
||||
)
|
||||
|
||||
self.blocks2 = nn.Sequential(
|
||||
*[
|
||||
LCNetV3Block(
|
||||
in_channels=make_divisible(in_c * scale),
|
||||
out_channels=make_divisible(out_c * scale),
|
||||
dw_size=k,
|
||||
stride=s,
|
||||
use_se=se,
|
||||
conv_kxk_num=conv_kxk_num,
|
||||
lr_mult=self.lr_mult_list[1],
|
||||
lab_lr=lab_lr,
|
||||
)
|
||||
for i, (k, in_c, out_c, s, se) in enumerate(self.net_config["blocks2"])
|
||||
]
|
||||
)
|
||||
|
||||
self.blocks3 = nn.Sequential(
|
||||
*[
|
||||
LCNetV3Block(
|
||||
in_channels=make_divisible(in_c * scale),
|
||||
out_channels=make_divisible(out_c * scale),
|
||||
dw_size=k,
|
||||
stride=s,
|
||||
use_se=se,
|
||||
conv_kxk_num=conv_kxk_num,
|
||||
lr_mult=self.lr_mult_list[2],
|
||||
lab_lr=lab_lr,
|
||||
)
|
||||
for i, (k, in_c, out_c, s, se) in enumerate(self.net_config["blocks3"])
|
||||
]
|
||||
)
|
||||
|
||||
self.blocks4 = nn.Sequential(
|
||||
*[
|
||||
LCNetV3Block(
|
||||
in_channels=make_divisible(in_c * scale),
|
||||
out_channels=make_divisible(out_c * scale),
|
||||
dw_size=k,
|
||||
stride=s,
|
||||
use_se=se,
|
||||
conv_kxk_num=conv_kxk_num,
|
||||
lr_mult=self.lr_mult_list[3],
|
||||
lab_lr=lab_lr,
|
||||
)
|
||||
for i, (k, in_c, out_c, s, se) in enumerate(self.net_config["blocks4"])
|
||||
]
|
||||
)
|
||||
|
||||
self.blocks5 = nn.Sequential(
|
||||
*[
|
||||
LCNetV3Block(
|
||||
in_channels=make_divisible(in_c * scale),
|
||||
out_channels=make_divisible(out_c * scale),
|
||||
dw_size=k,
|
||||
stride=s,
|
||||
use_se=se,
|
||||
conv_kxk_num=conv_kxk_num,
|
||||
lr_mult=self.lr_mult_list[4],
|
||||
lab_lr=lab_lr,
|
||||
)
|
||||
for i, (k, in_c, out_c, s, se) in enumerate(self.net_config["blocks5"])
|
||||
]
|
||||
)
|
||||
|
||||
self.blocks6 = nn.Sequential(
|
||||
*[
|
||||
LCNetV3Block(
|
||||
in_channels=make_divisible(in_c * scale),
|
||||
out_channels=make_divisible(out_c * scale),
|
||||
dw_size=k,
|
||||
stride=s,
|
||||
use_se=se,
|
||||
conv_kxk_num=conv_kxk_num,
|
||||
lr_mult=self.lr_mult_list[5],
|
||||
lab_lr=lab_lr,
|
||||
)
|
||||
for i, (k, in_c, out_c, s, se) in enumerate(self.net_config["blocks6"])
|
||||
]
|
||||
)
|
||||
self.out_channels = make_divisible(512 * scale)
|
||||
|
||||
if self.det:
|
||||
mv_c = [16, 24, 56, 480]
|
||||
self.out_channels = [
|
||||
make_divisible(self.net_config["blocks3"][-1][2] * scale),
|
||||
make_divisible(self.net_config["blocks4"][-1][2] * scale),
|
||||
make_divisible(self.net_config["blocks5"][-1][2] * scale),
|
||||
make_divisible(self.net_config["blocks6"][-1][2] * scale),
|
||||
]
|
||||
|
||||
self.layer_list = nn.LayerList(
|
||||
[
|
||||
nn.Conv2D(self.out_channels[0], int(mv_c[0] * scale), 1, 1, 0),
|
||||
nn.Conv2D(self.out_channels[1], int(mv_c[1] * scale), 1, 1, 0),
|
||||
nn.Conv2D(self.out_channels[2], int(mv_c[2] * scale), 1, 1, 0),
|
||||
nn.Conv2D(self.out_channels[3], int(mv_c[3] * scale), 1, 1, 0),
|
||||
]
|
||||
)
|
||||
self.out_channels = [
|
||||
int(mv_c[0] * scale),
|
||||
int(mv_c[1] * scale),
|
||||
int(mv_c[2] * scale),
|
||||
int(mv_c[3] * scale),
|
||||
]
|
||||
|
||||
def forward(self, x):
|
||||
out_list = []
|
||||
x = self.conv1(x)
|
||||
|
||||
x = self.blocks2(x)
|
||||
x = self.blocks3(x)
|
||||
out_list.append(x)
|
||||
x = self.blocks4(x)
|
||||
out_list.append(x)
|
||||
x = self.blocks5(x)
|
||||
out_list.append(x)
|
||||
x = self.blocks6(x)
|
||||
out_list.append(x)
|
||||
|
||||
if self.det:
|
||||
out_list[0] = self.layer_list[0](out_list[0])
|
||||
out_list[1] = self.layer_list[1](out_list[1])
|
||||
out_list[2] = self.layer_list[2](out_list[2])
|
||||
out_list[3] = self.layer_list[3](out_list[3])
|
||||
return out_list
|
||||
|
||||
if self.training:
|
||||
x = F.adaptive_avg_pool2d(x, [1, 40])
|
||||
else:
|
||||
x = F.avg_pool2d(x, [3, 2])
|
||||
return x
|
||||
605
ppocr/modeling/backbones/rec_micronet.py
Normal file
605
ppocr/modeling/backbones/rec_micronet.py
Normal file
@@ -0,0 +1,605 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/liyunsheng13/micronet/blob/main/backbone/micronet.py
|
||||
https://github.com/liyunsheng13/micronet/blob/main/backbone/activation.py
|
||||
"""
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
|
||||
from ppocr.modeling.backbones.det_mobilenet_v3 import make_divisible
|
||||
|
||||
M0_cfgs = [
|
||||
# s, n, c, ks, c1, c2, g1, g2, c3, g3, g4, y1, y2, y3, r
|
||||
[2, 1, 8, 3, 2, 2, 0, 4, 8, 2, 2, 2, 0, 1, 1],
|
||||
[2, 1, 12, 3, 2, 2, 0, 8, 12, 4, 4, 2, 2, 1, 1],
|
||||
[2, 1, 16, 5, 2, 2, 0, 12, 16, 4, 4, 2, 2, 1, 1],
|
||||
[1, 1, 32, 5, 1, 4, 4, 4, 32, 4, 4, 2, 2, 1, 1],
|
||||
[2, 1, 64, 5, 1, 4, 8, 8, 64, 8, 8, 2, 2, 1, 1],
|
||||
[1, 1, 96, 3, 1, 4, 8, 8, 96, 8, 8, 2, 2, 1, 2],
|
||||
[1, 1, 384, 3, 1, 4, 12, 12, 0, 0, 0, 2, 2, 1, 2],
|
||||
]
|
||||
M1_cfgs = [
|
||||
# s, n, c, ks, c1, c2, g1, g2, c3, g3, g4
|
||||
[2, 1, 8, 3, 2, 2, 0, 6, 8, 2, 2, 2, 0, 1, 1],
|
||||
[2, 1, 16, 3, 2, 2, 0, 8, 16, 4, 4, 2, 2, 1, 1],
|
||||
[2, 1, 16, 5, 2, 2, 0, 16, 16, 4, 4, 2, 2, 1, 1],
|
||||
[1, 1, 32, 5, 1, 6, 4, 4, 32, 4, 4, 2, 2, 1, 1],
|
||||
[2, 1, 64, 5, 1, 6, 8, 8, 64, 8, 8, 2, 2, 1, 1],
|
||||
[1, 1, 96, 3, 1, 6, 8, 8, 96, 8, 8, 2, 2, 1, 2],
|
||||
[1, 1, 576, 3, 1, 6, 12, 12, 0, 0, 0, 2, 2, 1, 2],
|
||||
]
|
||||
M2_cfgs = [
|
||||
# s, n, c, ks, c1, c2, g1, g2, c3, g3, g4
|
||||
[2, 1, 12, 3, 2, 2, 0, 8, 12, 4, 4, 2, 0, 1, 1],
|
||||
[2, 1, 16, 3, 2, 2, 0, 12, 16, 4, 4, 2, 2, 1, 1],
|
||||
[1, 1, 24, 3, 2, 2, 0, 16, 24, 4, 4, 2, 2, 1, 1],
|
||||
[2, 1, 32, 5, 1, 6, 6, 6, 32, 4, 4, 2, 2, 1, 1],
|
||||
[1, 1, 32, 5, 1, 6, 8, 8, 32, 4, 4, 2, 2, 1, 2],
|
||||
[1, 1, 64, 5, 1, 6, 8, 8, 64, 8, 8, 2, 2, 1, 2],
|
||||
[2, 1, 96, 5, 1, 6, 8, 8, 96, 8, 8, 2, 2, 1, 2],
|
||||
[1, 1, 128, 3, 1, 6, 12, 12, 128, 8, 8, 2, 2, 1, 2],
|
||||
[1, 1, 768, 3, 1, 6, 16, 16, 0, 0, 0, 2, 2, 1, 2],
|
||||
]
|
||||
M3_cfgs = [
|
||||
# s, n, c, ks, c1, c2, g1, g2, c3, g3, g4
|
||||
[2, 1, 16, 3, 2, 2, 0, 12, 16, 4, 4, 0, 2, 0, 1],
|
||||
[2, 1, 24, 3, 2, 2, 0, 16, 24, 4, 4, 0, 2, 0, 1],
|
||||
[1, 1, 24, 3, 2, 2, 0, 24, 24, 4, 4, 0, 2, 0, 1],
|
||||
[2, 1, 32, 5, 1, 6, 6, 6, 32, 4, 4, 0, 2, 0, 1],
|
||||
[1, 1, 32, 5, 1, 6, 8, 8, 32, 4, 4, 0, 2, 0, 2],
|
||||
[1, 1, 64, 5, 1, 6, 8, 8, 48, 8, 8, 0, 2, 0, 2],
|
||||
[1, 1, 80, 5, 1, 6, 8, 8, 80, 8, 8, 0, 2, 0, 2],
|
||||
[1, 1, 80, 5, 1, 6, 10, 10, 80, 8, 8, 0, 2, 0, 2],
|
||||
[1, 1, 120, 5, 1, 6, 10, 10, 120, 10, 10, 0, 2, 0, 2],
|
||||
[1, 1, 120, 5, 1, 6, 12, 12, 120, 10, 10, 0, 2, 0, 2],
|
||||
[1, 1, 144, 3, 1, 6, 12, 12, 144, 12, 12, 0, 2, 0, 2],
|
||||
[1, 1, 432, 3, 1, 3, 12, 12, 0, 0, 0, 0, 2, 0, 2],
|
||||
]
|
||||
|
||||
|
||||
def get_micronet_config(mode):
|
||||
return eval(mode + "_cfgs")
|
||||
|
||||
|
||||
class MaxGroupPooling(nn.Layer):
|
||||
def __init__(self, channel_per_group=2):
|
||||
super(MaxGroupPooling, self).__init__()
|
||||
self.channel_per_group = channel_per_group
|
||||
|
||||
def forward(self, x):
|
||||
if self.channel_per_group == 1:
|
||||
return x
|
||||
# max op
|
||||
b, c, h, w = x.shape
|
||||
|
||||
# reshape
|
||||
y = paddle.reshape(x, [b, c // self.channel_per_group, -1, h, w])
|
||||
out = paddle.max(y, axis=2)
|
||||
return out
|
||||
|
||||
|
||||
class SpatialSepConvSF(nn.Layer):
|
||||
def __init__(self, inp, oups, kernel_size, stride):
|
||||
super(SpatialSepConvSF, self).__init__()
|
||||
|
||||
oup1, oup2 = oups
|
||||
self.conv = nn.Sequential(
|
||||
nn.Conv2D(
|
||||
inp,
|
||||
oup1,
|
||||
(kernel_size, 1),
|
||||
(stride, 1),
|
||||
(kernel_size // 2, 0),
|
||||
bias_attr=False,
|
||||
groups=1,
|
||||
),
|
||||
nn.BatchNorm2D(oup1),
|
||||
nn.Conv2D(
|
||||
oup1,
|
||||
oup1 * oup2,
|
||||
(1, kernel_size),
|
||||
(1, stride),
|
||||
(0, kernel_size // 2),
|
||||
bias_attr=False,
|
||||
groups=oup1,
|
||||
),
|
||||
nn.BatchNorm2D(oup1 * oup2),
|
||||
ChannelShuffle(oup1),
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
out = self.conv(x)
|
||||
return out
|
||||
|
||||
|
||||
class ChannelShuffle(nn.Layer):
|
||||
def __init__(self, groups):
|
||||
super(ChannelShuffle, self).__init__()
|
||||
self.groups = groups
|
||||
|
||||
def forward(self, x):
|
||||
b, c, h, w = x.shape
|
||||
|
||||
channels_per_group = c // self.groups
|
||||
|
||||
# reshape
|
||||
x = paddle.reshape(x, [b, self.groups, channels_per_group, h, w])
|
||||
|
||||
x = paddle.transpose(x, (0, 2, 1, 3, 4))
|
||||
out = paddle.reshape(x, [b, -1, h, w])
|
||||
|
||||
return out
|
||||
|
||||
|
||||
class StemLayer(nn.Layer):
|
||||
def __init__(self, inp, oup, stride, groups=(4, 4)):
|
||||
super(StemLayer, self).__init__()
|
||||
|
||||
g1, g2 = groups
|
||||
self.stem = nn.Sequential(
|
||||
SpatialSepConvSF(inp, groups, 3, stride),
|
||||
MaxGroupPooling(2) if g1 * g2 == 2 * oup else nn.ReLU6(),
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
out = self.stem(x)
|
||||
return out
|
||||
|
||||
|
||||
class DepthSpatialSepConv(nn.Layer):
|
||||
def __init__(self, inp, expand, kernel_size, stride):
|
||||
super(DepthSpatialSepConv, self).__init__()
|
||||
|
||||
exp1, exp2 = expand
|
||||
|
||||
hidden_dim = inp * exp1
|
||||
oup = inp * exp1 * exp2
|
||||
|
||||
self.conv = nn.Sequential(
|
||||
nn.Conv2D(
|
||||
inp,
|
||||
inp * exp1,
|
||||
(kernel_size, 1),
|
||||
(stride, 1),
|
||||
(kernel_size // 2, 0),
|
||||
bias_attr=False,
|
||||
groups=inp,
|
||||
),
|
||||
nn.BatchNorm2D(inp * exp1),
|
||||
nn.Conv2D(
|
||||
hidden_dim,
|
||||
oup,
|
||||
(1, kernel_size),
|
||||
1,
|
||||
(0, kernel_size // 2),
|
||||
bias_attr=False,
|
||||
groups=hidden_dim,
|
||||
),
|
||||
nn.BatchNorm2D(oup),
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.conv(x)
|
||||
return x
|
||||
|
||||
|
||||
class GroupConv(nn.Layer):
|
||||
def __init__(self, inp, oup, groups=2):
|
||||
super(GroupConv, self).__init__()
|
||||
self.inp = inp
|
||||
self.oup = oup
|
||||
self.groups = groups
|
||||
self.conv = nn.Sequential(
|
||||
nn.Conv2D(inp, oup, 1, 1, 0, bias_attr=False, groups=self.groups[0]),
|
||||
nn.BatchNorm2D(oup),
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.conv(x)
|
||||
return x
|
||||
|
||||
|
||||
class DepthConv(nn.Layer):
|
||||
def __init__(self, inp, oup, kernel_size, stride):
|
||||
super(DepthConv, self).__init__()
|
||||
self.conv = nn.Sequential(
|
||||
nn.Conv2D(
|
||||
inp,
|
||||
oup,
|
||||
kernel_size,
|
||||
stride,
|
||||
kernel_size // 2,
|
||||
bias_attr=False,
|
||||
groups=inp,
|
||||
),
|
||||
nn.BatchNorm2D(oup),
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
out = self.conv(x)
|
||||
return out
|
||||
|
||||
|
||||
class DYShiftMax(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
inp,
|
||||
oup,
|
||||
reduction=4,
|
||||
act_max=1.0,
|
||||
act_relu=True,
|
||||
init_a=[0.0, 0.0],
|
||||
init_b=[0.0, 0.0],
|
||||
relu_before_pool=False,
|
||||
g=None,
|
||||
expansion=False,
|
||||
):
|
||||
super(DYShiftMax, self).__init__()
|
||||
self.oup = oup
|
||||
self.act_max = act_max * 2
|
||||
self.act_relu = act_relu
|
||||
self.avg_pool = nn.Sequential(
|
||||
nn.ReLU() if relu_before_pool == True else nn.Sequential(),
|
||||
nn.AdaptiveAvgPool2D(1),
|
||||
)
|
||||
|
||||
self.exp = 4 if act_relu else 2
|
||||
self.init_a = init_a
|
||||
self.init_b = init_b
|
||||
|
||||
# determine squeeze
|
||||
squeeze = make_divisible(inp // reduction, 4)
|
||||
if squeeze < 4:
|
||||
squeeze = 4
|
||||
|
||||
self.fc = nn.Sequential(
|
||||
nn.Linear(inp, squeeze),
|
||||
nn.ReLU(),
|
||||
nn.Linear(squeeze, oup * self.exp),
|
||||
nn.Hardsigmoid(),
|
||||
)
|
||||
|
||||
if g is None:
|
||||
g = 1
|
||||
self.g = g[1]
|
||||
if self.g != 1 and expansion:
|
||||
self.g = inp // self.g
|
||||
|
||||
self.gc = inp // self.g
|
||||
index = paddle.to_tensor([range(inp)])
|
||||
index = paddle.reshape(index, [1, inp, 1, 1])
|
||||
index = paddle.reshape(index, [1, self.g, self.gc, 1, 1])
|
||||
indexgs = paddle.split(index, [1, self.g - 1], axis=1)
|
||||
indexgs = paddle.concat((indexgs[1], indexgs[0]), axis=1)
|
||||
indexes = paddle.split(indexgs, [1, self.gc - 1], axis=2)
|
||||
indexes = paddle.concat((indexes[1], indexes[0]), axis=2)
|
||||
self.index = paddle.reshape(indexes, [inp])
|
||||
self.expansion = expansion
|
||||
|
||||
def forward(self, x):
|
||||
x_in = x
|
||||
x_out = x
|
||||
|
||||
b, c, _, _ = x_in.shape
|
||||
y = self.avg_pool(x_in)
|
||||
y = paddle.reshape(y, [b, c])
|
||||
y = self.fc(y)
|
||||
y = paddle.reshape(y, [b, self.oup * self.exp, 1, 1])
|
||||
y = (y - 0.5) * self.act_max
|
||||
|
||||
n2, c2, h2, w2 = x_out.shape
|
||||
x2 = paddle.to_tensor(x_out.numpy()[:, self.index.numpy(), :, :])
|
||||
|
||||
if self.exp == 4:
|
||||
temp = y.shape
|
||||
a1, b1, a2, b2 = paddle.split(y, temp[1] // self.oup, axis=1)
|
||||
|
||||
a1 = a1 + self.init_a[0]
|
||||
a2 = a2 + self.init_a[1]
|
||||
|
||||
b1 = b1 + self.init_b[0]
|
||||
b2 = b2 + self.init_b[1]
|
||||
|
||||
z1 = x_out * a1 + x2 * b1
|
||||
z2 = x_out * a2 + x2 * b2
|
||||
|
||||
out = paddle.maximum(z1, z2)
|
||||
|
||||
elif self.exp == 2:
|
||||
temp = y.shape
|
||||
a1, b1 = paddle.split(y, temp[1] // self.oup, axis=1)
|
||||
a1 = a1 + self.init_a[0]
|
||||
b1 = b1 + self.init_b[0]
|
||||
out = x_out * a1 + x2 * b1
|
||||
|
||||
return out
|
||||
|
||||
|
||||
class DYMicroBlock(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
inp,
|
||||
oup,
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
ch_exp=(2, 2),
|
||||
ch_per_group=4,
|
||||
groups_1x1=(1, 1),
|
||||
depthsep=True,
|
||||
shuffle=False,
|
||||
activation_cfg=None,
|
||||
):
|
||||
super(DYMicroBlock, self).__init__()
|
||||
|
||||
self.identity = stride == 1 and inp == oup
|
||||
|
||||
y1, y2, y3 = activation_cfg["dy"]
|
||||
act_reduction = 8 * activation_cfg["ratio"]
|
||||
init_a = activation_cfg["init_a"]
|
||||
init_b = activation_cfg["init_b"]
|
||||
|
||||
t1 = ch_exp
|
||||
gs1 = ch_per_group
|
||||
hidden_fft, g1, g2 = groups_1x1
|
||||
hidden_dim2 = inp * t1[0] * t1[1]
|
||||
|
||||
if gs1[0] == 0:
|
||||
self.layers = nn.Sequential(
|
||||
DepthSpatialSepConv(inp, t1, kernel_size, stride),
|
||||
(
|
||||
DYShiftMax(
|
||||
hidden_dim2,
|
||||
hidden_dim2,
|
||||
act_max=2.0,
|
||||
act_relu=True if y2 == 2 else False,
|
||||
init_a=init_a,
|
||||
reduction=act_reduction,
|
||||
init_b=init_b,
|
||||
g=gs1,
|
||||
expansion=False,
|
||||
)
|
||||
if y2 > 0
|
||||
else nn.ReLU6()
|
||||
),
|
||||
ChannelShuffle(gs1[1]) if shuffle else nn.Sequential(),
|
||||
(
|
||||
ChannelShuffle(hidden_dim2 // 2)
|
||||
if shuffle and y2 != 0
|
||||
else nn.Sequential()
|
||||
),
|
||||
GroupConv(hidden_dim2, oup, (g1, g2)),
|
||||
(
|
||||
DYShiftMax(
|
||||
oup,
|
||||
oup,
|
||||
act_max=2.0,
|
||||
act_relu=False,
|
||||
init_a=[1.0, 0.0],
|
||||
reduction=act_reduction // 2,
|
||||
init_b=[0.0, 0.0],
|
||||
g=(g1, g2),
|
||||
expansion=False,
|
||||
)
|
||||
if y3 > 0
|
||||
else nn.Sequential()
|
||||
),
|
||||
ChannelShuffle(g2) if shuffle else nn.Sequential(),
|
||||
(
|
||||
ChannelShuffle(oup // 2)
|
||||
if shuffle and oup % 2 == 0 and y3 != 0
|
||||
else nn.Sequential()
|
||||
),
|
||||
)
|
||||
elif g2 == 0:
|
||||
self.layers = nn.Sequential(
|
||||
GroupConv(inp, hidden_dim2, gs1),
|
||||
(
|
||||
DYShiftMax(
|
||||
hidden_dim2,
|
||||
hidden_dim2,
|
||||
act_max=2.0,
|
||||
act_relu=False,
|
||||
init_a=[1.0, 0.0],
|
||||
reduction=act_reduction,
|
||||
init_b=[0.0, 0.0],
|
||||
g=gs1,
|
||||
expansion=False,
|
||||
)
|
||||
if y3 > 0
|
||||
else nn.Sequential()
|
||||
),
|
||||
)
|
||||
else:
|
||||
self.layers = nn.Sequential(
|
||||
GroupConv(inp, hidden_dim2, gs1),
|
||||
(
|
||||
DYShiftMax(
|
||||
hidden_dim2,
|
||||
hidden_dim2,
|
||||
act_max=2.0,
|
||||
act_relu=True if y1 == 2 else False,
|
||||
init_a=init_a,
|
||||
reduction=act_reduction,
|
||||
init_b=init_b,
|
||||
g=gs1,
|
||||
expansion=False,
|
||||
)
|
||||
if y1 > 0
|
||||
else nn.ReLU6()
|
||||
),
|
||||
ChannelShuffle(gs1[1]) if shuffle else nn.Sequential(),
|
||||
(
|
||||
DepthSpatialSepConv(hidden_dim2, (1, 1), kernel_size, stride)
|
||||
if depthsep
|
||||
else DepthConv(hidden_dim2, hidden_dim2, kernel_size, stride)
|
||||
),
|
||||
nn.Sequential(),
|
||||
(
|
||||
DYShiftMax(
|
||||
hidden_dim2,
|
||||
hidden_dim2,
|
||||
act_max=2.0,
|
||||
act_relu=True if y2 == 2 else False,
|
||||
init_a=init_a,
|
||||
reduction=act_reduction,
|
||||
init_b=init_b,
|
||||
g=gs1,
|
||||
expansion=True,
|
||||
)
|
||||
if y2 > 0
|
||||
else nn.ReLU6()
|
||||
),
|
||||
(
|
||||
ChannelShuffle(hidden_dim2 // 4)
|
||||
if shuffle and y1 != 0 and y2 != 0
|
||||
else (
|
||||
nn.Sequential()
|
||||
if y1 == 0 and y2 == 0
|
||||
else ChannelShuffle(hidden_dim2 // 2)
|
||||
)
|
||||
),
|
||||
GroupConv(hidden_dim2, oup, (g1, g2)),
|
||||
(
|
||||
DYShiftMax(
|
||||
oup,
|
||||
oup,
|
||||
act_max=2.0,
|
||||
act_relu=False,
|
||||
init_a=[1.0, 0.0],
|
||||
reduction=(
|
||||
act_reduction // 2 if oup < hidden_dim2 else act_reduction
|
||||
),
|
||||
init_b=[0.0, 0.0],
|
||||
g=(g1, g2),
|
||||
expansion=False,
|
||||
)
|
||||
if y3 > 0
|
||||
else nn.Sequential()
|
||||
),
|
||||
ChannelShuffle(g2) if shuffle else nn.Sequential(),
|
||||
ChannelShuffle(oup // 2) if shuffle and y3 != 0 else nn.Sequential(),
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
identity = x
|
||||
out = self.layers(x)
|
||||
|
||||
if self.identity:
|
||||
out = out + identity
|
||||
|
||||
return out
|
||||
|
||||
|
||||
class MicroNet(nn.Layer):
|
||||
"""
|
||||
the MicroNet backbone network for recognition module.
|
||||
Args:
|
||||
mode(str): {'M0', 'M1', 'M2', 'M3'}
|
||||
Four models are proposed based on four different computational costs (4M, 6M, 12M, 21M MAdds)
|
||||
Default: 'M3'.
|
||||
"""
|
||||
|
||||
def __init__(self, mode="M3", **kwargs):
|
||||
super(MicroNet, self).__init__()
|
||||
|
||||
self.cfgs = get_micronet_config(mode)
|
||||
|
||||
activation_cfg = {}
|
||||
if mode == "M0":
|
||||
input_channel = 4
|
||||
stem_groups = 2, 2
|
||||
out_ch = 384
|
||||
activation_cfg["init_a"] = 1.0, 1.0
|
||||
activation_cfg["init_b"] = 0.0, 0.0
|
||||
elif mode == "M1":
|
||||
input_channel = 6
|
||||
stem_groups = 3, 2
|
||||
out_ch = 576
|
||||
activation_cfg["init_a"] = 1.0, 1.0
|
||||
activation_cfg["init_b"] = 0.0, 0.0
|
||||
elif mode == "M2":
|
||||
input_channel = 8
|
||||
stem_groups = 4, 2
|
||||
out_ch = 768
|
||||
activation_cfg["init_a"] = 1.0, 1.0
|
||||
activation_cfg["init_b"] = 0.0, 0.0
|
||||
elif mode == "M3":
|
||||
input_channel = 12
|
||||
stem_groups = 4, 3
|
||||
out_ch = 432
|
||||
activation_cfg["init_a"] = 1.0, 0.5
|
||||
activation_cfg["init_b"] = 0.0, 0.5
|
||||
else:
|
||||
raise NotImplementedError("mode[" + mode + "_model] is not implemented!")
|
||||
|
||||
layers = [StemLayer(3, input_channel, stride=2, groups=stem_groups)]
|
||||
|
||||
for idx, val in enumerate(self.cfgs):
|
||||
s, n, c, ks, c1, c2, g1, g2, c3, g3, g4, y1, y2, y3, r = val
|
||||
|
||||
t1 = (c1, c2)
|
||||
gs1 = (g1, g2)
|
||||
gs2 = (c3, g3, g4)
|
||||
activation_cfg["dy"] = [y1, y2, y3]
|
||||
activation_cfg["ratio"] = r
|
||||
|
||||
output_channel = c
|
||||
layers.append(
|
||||
DYMicroBlock(
|
||||
input_channel,
|
||||
output_channel,
|
||||
kernel_size=ks,
|
||||
stride=s,
|
||||
ch_exp=t1,
|
||||
ch_per_group=gs1,
|
||||
groups_1x1=gs2,
|
||||
depthsep=True,
|
||||
shuffle=True,
|
||||
activation_cfg=activation_cfg,
|
||||
)
|
||||
)
|
||||
input_channel = output_channel
|
||||
for i in range(1, n):
|
||||
layers.append(
|
||||
DYMicroBlock(
|
||||
input_channel,
|
||||
output_channel,
|
||||
kernel_size=ks,
|
||||
stride=1,
|
||||
ch_exp=t1,
|
||||
ch_per_group=gs1,
|
||||
groups_1x1=gs2,
|
||||
depthsep=True,
|
||||
shuffle=True,
|
||||
activation_cfg=activation_cfg,
|
||||
)
|
||||
)
|
||||
input_channel = output_channel
|
||||
self.features = nn.Sequential(*layers)
|
||||
|
||||
self.pool = nn.MaxPool2D(kernel_size=2, stride=2, padding=0)
|
||||
|
||||
self.out_channels = make_divisible(out_ch)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.features(x)
|
||||
x = self.pool(x)
|
||||
return x
|
||||
156
ppocr/modeling/backbones/rec_mobilenet_v3.py
Normal file
156
ppocr/modeling/backbones/rec_mobilenet_v3.py
Normal file
@@ -0,0 +1,156 @@
|
||||
# copyright (c) 2020 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from paddle import nn
|
||||
|
||||
from ppocr.modeling.backbones.det_mobilenet_v3 import (
|
||||
ResidualUnit,
|
||||
ConvBNLayer,
|
||||
make_divisible,
|
||||
)
|
||||
|
||||
__all__ = ["MobileNetV3"]
|
||||
|
||||
|
||||
class MobileNetV3(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels=3,
|
||||
model_name="small",
|
||||
scale=0.5,
|
||||
large_stride=None,
|
||||
small_stride=None,
|
||||
disable_se=False,
|
||||
**kwargs,
|
||||
):
|
||||
super(MobileNetV3, self).__init__()
|
||||
self.disable_se = disable_se
|
||||
if small_stride is None:
|
||||
small_stride = [2, 2, 2, 2]
|
||||
if large_stride is None:
|
||||
large_stride = [1, 2, 2, 2]
|
||||
|
||||
assert isinstance(
|
||||
large_stride, list
|
||||
), "large_stride type must " "be list but got {}".format(type(large_stride))
|
||||
assert isinstance(
|
||||
small_stride, list
|
||||
), "small_stride type must " "be list but got {}".format(type(small_stride))
|
||||
assert (
|
||||
len(large_stride) == 4
|
||||
), "large_stride length must be " "4 but got {}".format(len(large_stride))
|
||||
assert (
|
||||
len(small_stride) == 4
|
||||
), "small_stride length must be " "4 but got {}".format(len(small_stride))
|
||||
|
||||
if model_name == "large":
|
||||
cfg = [
|
||||
# k, exp, c, se, nl, s,
|
||||
[3, 16, 16, False, "relu", large_stride[0]],
|
||||
[3, 64, 24, False, "relu", (large_stride[1], 1)],
|
||||
[3, 72, 24, False, "relu", 1],
|
||||
[5, 72, 40, True, "relu", (large_stride[2], 1)],
|
||||
[5, 120, 40, True, "relu", 1],
|
||||
[5, 120, 40, True, "relu", 1],
|
||||
[3, 240, 80, False, "hardswish", 1],
|
||||
[3, 200, 80, False, "hardswish", 1],
|
||||
[3, 184, 80, False, "hardswish", 1],
|
||||
[3, 184, 80, False, "hardswish", 1],
|
||||
[3, 480, 112, True, "hardswish", 1],
|
||||
[3, 672, 112, True, "hardswish", 1],
|
||||
[5, 672, 160, True, "hardswish", (large_stride[3], 1)],
|
||||
[5, 960, 160, True, "hardswish", 1],
|
||||
[5, 960, 160, True, "hardswish", 1],
|
||||
]
|
||||
cls_ch_squeeze = 960
|
||||
elif model_name == "small":
|
||||
cfg = [
|
||||
# k, exp, c, se, nl, s,
|
||||
[3, 16, 16, True, "relu", (small_stride[0], 1)],
|
||||
[3, 72, 24, False, "relu", (small_stride[1], 1)],
|
||||
[3, 88, 24, False, "relu", 1],
|
||||
[5, 96, 40, True, "hardswish", (small_stride[2], 1)],
|
||||
[5, 240, 40, True, "hardswish", 1],
|
||||
[5, 240, 40, True, "hardswish", 1],
|
||||
[5, 120, 48, True, "hardswish", 1],
|
||||
[5, 144, 48, True, "hardswish", 1],
|
||||
[5, 288, 96, True, "hardswish", (small_stride[3], 1)],
|
||||
[5, 576, 96, True, "hardswish", 1],
|
||||
[5, 576, 96, True, "hardswish", 1],
|
||||
]
|
||||
cls_ch_squeeze = 576
|
||||
else:
|
||||
raise NotImplementedError(
|
||||
"mode[" + model_name + "_model] is not implemented!"
|
||||
)
|
||||
|
||||
supported_scale = [0.35, 0.5, 0.75, 1.0, 1.25]
|
||||
assert (
|
||||
scale in supported_scale
|
||||
), "supported scales are {} but input scale is {}".format(
|
||||
supported_scale, scale
|
||||
)
|
||||
|
||||
inplanes = 16
|
||||
# conv1
|
||||
self.conv1 = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=make_divisible(inplanes * scale),
|
||||
kernel_size=3,
|
||||
stride=2,
|
||||
padding=1,
|
||||
groups=1,
|
||||
if_act=True,
|
||||
act="hardswish",
|
||||
)
|
||||
i = 0
|
||||
block_list = []
|
||||
inplanes = make_divisible(inplanes * scale)
|
||||
for k, exp, c, se, nl, s in cfg:
|
||||
se = se and not self.disable_se
|
||||
block_list.append(
|
||||
ResidualUnit(
|
||||
in_channels=inplanes,
|
||||
mid_channels=make_divisible(scale * exp),
|
||||
out_channels=make_divisible(scale * c),
|
||||
kernel_size=k,
|
||||
stride=s,
|
||||
use_se=se,
|
||||
act=nl,
|
||||
)
|
||||
)
|
||||
inplanes = make_divisible(scale * c)
|
||||
i += 1
|
||||
self.blocks = nn.Sequential(*block_list)
|
||||
|
||||
self.conv2 = ConvBNLayer(
|
||||
in_channels=inplanes,
|
||||
out_channels=make_divisible(scale * cls_ch_squeeze),
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
padding=0,
|
||||
groups=1,
|
||||
if_act=True,
|
||||
act="hardswish",
|
||||
)
|
||||
|
||||
self.pool = nn.MaxPool2D(kernel_size=2, stride=2, padding=0)
|
||||
self.out_channels = make_divisible(scale * cls_ch_squeeze)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.conv1(x)
|
||||
x = self.blocks(x)
|
||||
x = self.conv2(x)
|
||||
x = self.pool(x)
|
||||
return x
|
||||
283
ppocr/modeling/backbones/rec_mv1_enhance.py
Normal file
283
ppocr/modeling/backbones/rec_mv1_enhance.py
Normal file
@@ -0,0 +1,283 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
# This code is refer from: https://github.com/PaddlePaddle/PaddleClas/blob/develop/ppcls/arch/backbone/legendary_models/pp_lcnet.py
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import math
|
||||
import numpy as np
|
||||
import paddle
|
||||
from paddle import ParamAttr, reshape, transpose
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
from paddle.nn import Conv2D, BatchNorm, Linear, Dropout
|
||||
from paddle.nn import AdaptiveAvgPool2D, MaxPool2D, AvgPool2D
|
||||
from paddle.nn.initializer import KaimingNormal
|
||||
from paddle.regularizer import L2Decay
|
||||
from paddle.nn.functional import hardswish, hardsigmoid
|
||||
|
||||
|
||||
class ConvBNLayer(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
num_channels,
|
||||
filter_size,
|
||||
num_filters,
|
||||
stride,
|
||||
padding,
|
||||
channels=None,
|
||||
num_groups=1,
|
||||
act="hard_swish",
|
||||
):
|
||||
super(ConvBNLayer, self).__init__()
|
||||
|
||||
self._conv = Conv2D(
|
||||
in_channels=num_channels,
|
||||
out_channels=num_filters,
|
||||
kernel_size=filter_size,
|
||||
stride=stride,
|
||||
padding=padding,
|
||||
groups=num_groups,
|
||||
weight_attr=ParamAttr(initializer=KaimingNormal()),
|
||||
bias_attr=False,
|
||||
)
|
||||
|
||||
self._batch_norm = BatchNorm(
|
||||
num_filters,
|
||||
act=act,
|
||||
param_attr=ParamAttr(regularizer=L2Decay(0.0)),
|
||||
bias_attr=ParamAttr(regularizer=L2Decay(0.0)),
|
||||
)
|
||||
|
||||
def forward(self, inputs):
|
||||
y = self._conv(inputs)
|
||||
y = self._batch_norm(y)
|
||||
return y
|
||||
|
||||
|
||||
class DepthwiseSeparable(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
num_channels,
|
||||
num_filters1,
|
||||
num_filters2,
|
||||
num_groups,
|
||||
stride,
|
||||
scale,
|
||||
dw_size=3,
|
||||
padding=1,
|
||||
use_se=False,
|
||||
):
|
||||
super(DepthwiseSeparable, self).__init__()
|
||||
self.use_se = use_se
|
||||
self._depthwise_conv = ConvBNLayer(
|
||||
num_channels=num_channels,
|
||||
num_filters=int(num_filters1 * scale),
|
||||
filter_size=dw_size,
|
||||
stride=stride,
|
||||
padding=padding,
|
||||
num_groups=int(num_groups * scale),
|
||||
)
|
||||
if use_se:
|
||||
self._se = SEModule(int(num_filters1 * scale))
|
||||
self._pointwise_conv = ConvBNLayer(
|
||||
num_channels=int(num_filters1 * scale),
|
||||
filter_size=1,
|
||||
num_filters=int(num_filters2 * scale),
|
||||
stride=1,
|
||||
padding=0,
|
||||
)
|
||||
|
||||
def forward(self, inputs):
|
||||
y = self._depthwise_conv(inputs)
|
||||
if self.use_se:
|
||||
y = self._se(y)
|
||||
y = self._pointwise_conv(y)
|
||||
return y
|
||||
|
||||
|
||||
class MobileNetV1Enhance(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels=3,
|
||||
scale=0.5,
|
||||
last_conv_stride=1,
|
||||
last_pool_type="max",
|
||||
last_pool_kernel_size=[3, 2],
|
||||
**kwargs,
|
||||
):
|
||||
super().__init__()
|
||||
self.scale = scale
|
||||
self.block_list = []
|
||||
|
||||
self.conv1 = ConvBNLayer(
|
||||
num_channels=3,
|
||||
filter_size=3,
|
||||
channels=3,
|
||||
num_filters=int(32 * scale),
|
||||
stride=2,
|
||||
padding=1,
|
||||
)
|
||||
|
||||
conv2_1 = DepthwiseSeparable(
|
||||
num_channels=int(32 * scale),
|
||||
num_filters1=32,
|
||||
num_filters2=64,
|
||||
num_groups=32,
|
||||
stride=1,
|
||||
scale=scale,
|
||||
)
|
||||
self.block_list.append(conv2_1)
|
||||
|
||||
conv2_2 = DepthwiseSeparable(
|
||||
num_channels=int(64 * scale),
|
||||
num_filters1=64,
|
||||
num_filters2=128,
|
||||
num_groups=64,
|
||||
stride=1,
|
||||
scale=scale,
|
||||
)
|
||||
self.block_list.append(conv2_2)
|
||||
|
||||
conv3_1 = DepthwiseSeparable(
|
||||
num_channels=int(128 * scale),
|
||||
num_filters1=128,
|
||||
num_filters2=128,
|
||||
num_groups=128,
|
||||
stride=1,
|
||||
scale=scale,
|
||||
)
|
||||
self.block_list.append(conv3_1)
|
||||
|
||||
conv3_2 = DepthwiseSeparable(
|
||||
num_channels=int(128 * scale),
|
||||
num_filters1=128,
|
||||
num_filters2=256,
|
||||
num_groups=128,
|
||||
stride=(2, 1),
|
||||
scale=scale,
|
||||
)
|
||||
self.block_list.append(conv3_2)
|
||||
|
||||
conv4_1 = DepthwiseSeparable(
|
||||
num_channels=int(256 * scale),
|
||||
num_filters1=256,
|
||||
num_filters2=256,
|
||||
num_groups=256,
|
||||
stride=1,
|
||||
scale=scale,
|
||||
)
|
||||
self.block_list.append(conv4_1)
|
||||
|
||||
conv4_2 = DepthwiseSeparable(
|
||||
num_channels=int(256 * scale),
|
||||
num_filters1=256,
|
||||
num_filters2=512,
|
||||
num_groups=256,
|
||||
stride=(2, 1),
|
||||
scale=scale,
|
||||
)
|
||||
self.block_list.append(conv4_2)
|
||||
|
||||
for _ in range(5):
|
||||
conv5 = DepthwiseSeparable(
|
||||
num_channels=int(512 * scale),
|
||||
num_filters1=512,
|
||||
num_filters2=512,
|
||||
num_groups=512,
|
||||
stride=1,
|
||||
dw_size=5,
|
||||
padding=2,
|
||||
scale=scale,
|
||||
use_se=False,
|
||||
)
|
||||
self.block_list.append(conv5)
|
||||
|
||||
conv5_6 = DepthwiseSeparable(
|
||||
num_channels=int(512 * scale),
|
||||
num_filters1=512,
|
||||
num_filters2=1024,
|
||||
num_groups=512,
|
||||
stride=(2, 1),
|
||||
dw_size=5,
|
||||
padding=2,
|
||||
scale=scale,
|
||||
use_se=True,
|
||||
)
|
||||
self.block_list.append(conv5_6)
|
||||
|
||||
conv6 = DepthwiseSeparable(
|
||||
num_channels=int(1024 * scale),
|
||||
num_filters1=1024,
|
||||
num_filters2=1024,
|
||||
num_groups=1024,
|
||||
stride=last_conv_stride,
|
||||
dw_size=5,
|
||||
padding=2,
|
||||
use_se=True,
|
||||
scale=scale,
|
||||
)
|
||||
self.block_list.append(conv6)
|
||||
|
||||
self.block_list = nn.Sequential(*self.block_list)
|
||||
if last_pool_type == "avg":
|
||||
self.pool = nn.AvgPool2D(
|
||||
kernel_size=last_pool_kernel_size,
|
||||
stride=last_pool_kernel_size,
|
||||
padding=0,
|
||||
)
|
||||
else:
|
||||
self.pool = nn.MaxPool2D(kernel_size=2, stride=2, padding=0)
|
||||
self.out_channels = int(1024 * scale)
|
||||
|
||||
def forward(self, inputs):
|
||||
y = self.conv1(inputs)
|
||||
y = self.block_list(y)
|
||||
y = self.pool(y)
|
||||
return y
|
||||
|
||||
|
||||
class SEModule(nn.Layer):
|
||||
def __init__(self, channel, reduction=4):
|
||||
super(SEModule, self).__init__()
|
||||
self.avg_pool = AdaptiveAvgPool2D(1)
|
||||
self.conv1 = Conv2D(
|
||||
in_channels=channel,
|
||||
out_channels=channel // reduction,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
padding=0,
|
||||
weight_attr=ParamAttr(),
|
||||
bias_attr=ParamAttr(),
|
||||
)
|
||||
self.conv2 = Conv2D(
|
||||
in_channels=channel // reduction,
|
||||
out_channels=channel,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
padding=0,
|
||||
weight_attr=ParamAttr(),
|
||||
bias_attr=ParamAttr(),
|
||||
)
|
||||
|
||||
def forward(self, inputs):
|
||||
outputs = self.avg_pool(inputs)
|
||||
outputs = self.conv1(outputs)
|
||||
outputs = F.relu(outputs)
|
||||
outputs = self.conv2(outputs)
|
||||
outputs = hardsigmoid(outputs)
|
||||
return paddle.multiply(x=inputs, y=outputs)
|
||||
47
ppocr/modeling/backbones/rec_nrtr_mtb.py
Normal file
47
ppocr/modeling/backbones/rec_nrtr_mtb.py
Normal file
@@ -0,0 +1,47 @@
|
||||
# copyright (c) 2020 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from paddle import nn
|
||||
import paddle
|
||||
|
||||
|
||||
class MTB(nn.Layer):
|
||||
def __init__(self, cnn_num, in_channels):
|
||||
super(MTB, self).__init__()
|
||||
self.block = nn.Sequential()
|
||||
self.out_channels = in_channels
|
||||
self.cnn_num = cnn_num
|
||||
if self.cnn_num == 2:
|
||||
for i in range(self.cnn_num):
|
||||
self.block.add_sublayer(
|
||||
"conv_{}".format(i),
|
||||
nn.Conv2D(
|
||||
in_channels=in_channels if i == 0 else 32 * (2 ** (i - 1)),
|
||||
out_channels=32 * (2**i),
|
||||
kernel_size=3,
|
||||
stride=2,
|
||||
padding=1,
|
||||
),
|
||||
)
|
||||
self.block.add_sublayer("relu_{}".format(i), nn.ReLU())
|
||||
self.block.add_sublayer("bn_{}".format(i), nn.BatchNorm2D(32 * (2**i)))
|
||||
|
||||
def forward(self, images):
|
||||
x = self.block(images)
|
||||
if self.cnn_num == 2:
|
||||
# (b, w, h, c)
|
||||
x = paddle.transpose(x, [0, 3, 2, 1])
|
||||
x_shape = x.shape
|
||||
x = paddle.reshape(x, [x_shape[0], x_shape[1], x_shape[2] * x_shape[3]])
|
||||
return x
|
||||
1713
ppocr/modeling/backbones/rec_pphgnetv2.py
Normal file
1713
ppocr/modeling/backbones/rec_pphgnetv2.py
Normal file
File diff suppressed because it is too large
Load Diff
363
ppocr/modeling/backbones/rec_repvit.py
Normal file
363
ppocr/modeling/backbones/rec_repvit.py
Normal file
@@ -0,0 +1,363 @@
|
||||
# copyright (c) 2024 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/THU-MIG/RepViT
|
||||
"""
|
||||
|
||||
import paddle.nn as nn
|
||||
import paddle
|
||||
from paddle.nn.initializer import TruncatedNormal, Constant, Normal
|
||||
|
||||
trunc_normal_ = TruncatedNormal(std=0.02)
|
||||
normal_ = Normal
|
||||
zeros_ = Constant(value=0.0)
|
||||
ones_ = Constant(value=1.0)
|
||||
|
||||
|
||||
def _make_divisible(v, divisor, min_value=None):
|
||||
"""
|
||||
This function is taken from the original tf repo.
|
||||
It ensures that all layers have a channel number that is divisible by 8
|
||||
It can be seen here:
|
||||
https://github.com/tensorflow/models/blob/master/research/slim/nets/mobilenet/mobilenet.py
|
||||
:param v:
|
||||
:param divisor:
|
||||
:param min_value:
|
||||
:return:
|
||||
"""
|
||||
if min_value is None:
|
||||
min_value = divisor
|
||||
new_v = max(min_value, int(v + divisor / 2) // divisor * divisor)
|
||||
# Make sure that round down does not go down by more than 10%.
|
||||
if new_v < 0.9 * v:
|
||||
new_v += divisor
|
||||
return new_v
|
||||
|
||||
|
||||
# from timm.models.layers import SqueezeExcite
|
||||
|
||||
|
||||
def make_divisible(v, divisor=8, min_value=None, round_limit=0.9):
|
||||
min_value = min_value or divisor
|
||||
new_v = max(min_value, int(v + divisor / 2) // divisor * divisor)
|
||||
# Make sure that round down does not go down by more than 10%.
|
||||
if new_v < round_limit * v:
|
||||
new_v += divisor
|
||||
return new_v
|
||||
|
||||
|
||||
class SEModule(nn.Layer):
|
||||
"""SE Module as defined in original SE-Nets with a few additions
|
||||
Additions include:
|
||||
* divisor can be specified to keep channels % div == 0 (default: 8)
|
||||
* reduction channels can be specified directly by arg (if rd_channels is set)
|
||||
* reduction channels can be specified by float rd_ratio (default: 1/16)
|
||||
* global max pooling can be added to the squeeze aggregation
|
||||
* customizable activation, normalization, and gate layer
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
channels,
|
||||
rd_ratio=1.0 / 16,
|
||||
rd_channels=None,
|
||||
rd_divisor=8,
|
||||
act_layer=nn.ReLU,
|
||||
):
|
||||
super(SEModule, self).__init__()
|
||||
if not rd_channels:
|
||||
rd_channels = make_divisible(
|
||||
channels * rd_ratio, rd_divisor, round_limit=0.0
|
||||
)
|
||||
self.fc1 = nn.Conv2D(channels, rd_channels, kernel_size=1, bias_attr=True)
|
||||
self.act = act_layer()
|
||||
self.fc2 = nn.Conv2D(rd_channels, channels, kernel_size=1, bias_attr=True)
|
||||
|
||||
def forward(self, x):
|
||||
x_se = x.mean((2, 3), keepdim=True)
|
||||
x_se = self.fc1(x_se)
|
||||
x_se = self.act(x_se)
|
||||
x_se = self.fc2(x_se)
|
||||
return x * nn.functional.sigmoid(x_se)
|
||||
|
||||
|
||||
class Conv2D_BN(nn.Sequential):
|
||||
def __init__(
|
||||
self,
|
||||
a,
|
||||
b,
|
||||
ks=1,
|
||||
stride=1,
|
||||
pad=0,
|
||||
dilation=1,
|
||||
groups=1,
|
||||
bn_weight_init=1,
|
||||
resolution=-10000,
|
||||
):
|
||||
super().__init__()
|
||||
self.add_sublayer(
|
||||
"c", nn.Conv2D(a, b, ks, stride, pad, dilation, groups, bias_attr=False)
|
||||
)
|
||||
self.add_sublayer("bn", nn.BatchNorm2D(b))
|
||||
if bn_weight_init == 1:
|
||||
ones_(self.bn.weight)
|
||||
else:
|
||||
zeros_(self.bn.weight)
|
||||
zeros_(self.bn.bias)
|
||||
|
||||
@paddle.no_grad()
|
||||
def fuse(self):
|
||||
c, bn = self.c, self.bn
|
||||
w = bn.weight / (bn._variance + bn._epsilon) ** 0.5
|
||||
w = c.weight * w[:, None, None, None]
|
||||
b = bn.bias - bn._mean * bn.weight / (bn._variance + bn._epsilon) ** 0.5
|
||||
m = nn.Conv2D(
|
||||
w.shape[1] * self.c._groups,
|
||||
w.shape[0],
|
||||
w.shape[2:],
|
||||
stride=self.c._stride,
|
||||
padding=self.c._padding,
|
||||
dilation=self.c._dilation,
|
||||
groups=self.c._groups,
|
||||
)
|
||||
m.weight.set_value(w)
|
||||
m.bias.set_value(b)
|
||||
return m
|
||||
|
||||
|
||||
class Residual(nn.Layer):
|
||||
def __init__(self, m, drop=0.0):
|
||||
super().__init__()
|
||||
self.m = m
|
||||
self.drop = drop
|
||||
|
||||
def forward(self, x):
|
||||
if self.training and self.drop > 0:
|
||||
return (
|
||||
x
|
||||
+ self.m(x)
|
||||
* paddle.rand(x.size(0), 1, 1, 1)
|
||||
.ge_(self.drop)
|
||||
.div(1 - self.drop)
|
||||
.detach()
|
||||
)
|
||||
else:
|
||||
return x + self.m(x)
|
||||
|
||||
@paddle.no_grad()
|
||||
def fuse(self):
|
||||
if isinstance(self.m, Conv2D_BN):
|
||||
m = self.m.fuse()
|
||||
assert m._groups == m.in_channels
|
||||
identity = paddle.ones([m.weight.shape[0], m.weight.shape[1], 1, 1])
|
||||
identity = nn.functional.pad(identity, [1, 1, 1, 1])
|
||||
m.weight += identity
|
||||
return m
|
||||
elif isinstance(self.m, nn.Conv2D):
|
||||
m = self.m
|
||||
assert m._groups != m.in_channels
|
||||
identity = paddle.ones([m.weight.shape[0], m.weight.shape[1], 1, 1])
|
||||
identity = nn.functional.pad(identity, [1, 1, 1, 1])
|
||||
m.weight += identity
|
||||
return m
|
||||
else:
|
||||
return self
|
||||
|
||||
|
||||
class RepVGGDW(nn.Layer):
|
||||
def __init__(self, ed) -> None:
|
||||
super().__init__()
|
||||
self.conv = Conv2D_BN(ed, ed, 3, 1, 1, groups=ed)
|
||||
self.conv1 = nn.Conv2D(ed, ed, 1, 1, 0, groups=ed)
|
||||
self.dim = ed
|
||||
self.bn = nn.BatchNorm2D(ed)
|
||||
|
||||
def forward(self, x):
|
||||
return self.bn((self.conv(x) + self.conv1(x)) + x)
|
||||
|
||||
@paddle.no_grad()
|
||||
def fuse(self):
|
||||
conv = self.conv.fuse()
|
||||
conv1 = self.conv1
|
||||
|
||||
conv_w = conv.weight
|
||||
conv_b = conv.bias
|
||||
conv1_w = conv1.weight
|
||||
conv1_b = conv1.bias
|
||||
|
||||
conv1_w = nn.functional.pad(conv1_w, [1, 1, 1, 1])
|
||||
|
||||
identity = nn.functional.pad(
|
||||
paddle.ones([conv1_w.shape[0], conv1_w.shape[1], 1, 1]), [1, 1, 1, 1]
|
||||
)
|
||||
|
||||
final_conv_w = conv_w + conv1_w + identity
|
||||
final_conv_b = conv_b + conv1_b
|
||||
|
||||
conv.weight.set_value(final_conv_w)
|
||||
conv.bias.set_value(final_conv_b)
|
||||
|
||||
bn = self.bn
|
||||
w = bn.weight / (bn._variance + bn._epsilon) ** 0.5
|
||||
w = conv.weight * w[:, None, None, None]
|
||||
b = (
|
||||
bn.bias
|
||||
+ (conv.bias - bn._mean) * bn.weight / (bn._variance + bn._epsilon) ** 0.5
|
||||
)
|
||||
conv.weight.set_value(w)
|
||||
conv.bias.set_value(b)
|
||||
return conv
|
||||
|
||||
|
||||
class RepViTBlock(nn.Layer):
|
||||
def __init__(self, inp, hidden_dim, oup, kernel_size, stride, use_se, use_hs):
|
||||
super(RepViTBlock, self).__init__()
|
||||
|
||||
self.identity = stride == 1 and inp == oup
|
||||
assert hidden_dim == 2 * inp
|
||||
|
||||
if stride != 1:
|
||||
self.token_mixer = nn.Sequential(
|
||||
Conv2D_BN(
|
||||
inp, inp, kernel_size, stride, (kernel_size - 1) // 2, groups=inp
|
||||
),
|
||||
SEModule(inp, 0.25) if use_se else nn.Identity(),
|
||||
Conv2D_BN(inp, oup, ks=1, stride=1, pad=0),
|
||||
)
|
||||
self.channel_mixer = Residual(
|
||||
nn.Sequential(
|
||||
# pw
|
||||
Conv2D_BN(oup, 2 * oup, 1, 1, 0),
|
||||
nn.GELU() if use_hs else nn.GELU(),
|
||||
# pw-linear
|
||||
Conv2D_BN(2 * oup, oup, 1, 1, 0, bn_weight_init=0),
|
||||
)
|
||||
)
|
||||
else:
|
||||
assert self.identity
|
||||
self.token_mixer = nn.Sequential(
|
||||
RepVGGDW(inp),
|
||||
SEModule(inp, 0.25) if use_se else nn.Identity(),
|
||||
)
|
||||
self.channel_mixer = Residual(
|
||||
nn.Sequential(
|
||||
# pw
|
||||
Conv2D_BN(inp, hidden_dim, 1, 1, 0),
|
||||
nn.GELU() if use_hs else nn.GELU(),
|
||||
# pw-linear
|
||||
Conv2D_BN(hidden_dim, oup, 1, 1, 0, bn_weight_init=0),
|
||||
)
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
return self.channel_mixer(self.token_mixer(x))
|
||||
|
||||
|
||||
class RepViT(nn.Layer):
|
||||
def __init__(self, cfgs, in_channels=3, out_indices=None):
|
||||
super(RepViT, self).__init__()
|
||||
# setting of inverted residual blocks
|
||||
self.cfgs = cfgs
|
||||
|
||||
# building first layer
|
||||
input_channel = self.cfgs[0][2]
|
||||
patch_embed = nn.Sequential(
|
||||
Conv2D_BN(in_channels, input_channel // 2, 3, 2, 1),
|
||||
nn.GELU(),
|
||||
Conv2D_BN(input_channel // 2, input_channel, 3, 2, 1),
|
||||
)
|
||||
layers = [patch_embed]
|
||||
# building inverted residual blocks
|
||||
block = RepViTBlock
|
||||
for k, t, c, use_se, use_hs, s in self.cfgs:
|
||||
output_channel = _make_divisible(c, 8)
|
||||
exp_size = _make_divisible(input_channel * t, 8)
|
||||
layers.append(
|
||||
block(input_channel, exp_size, output_channel, k, s, use_se, use_hs)
|
||||
)
|
||||
input_channel = output_channel
|
||||
self.features = nn.LayerList(layers)
|
||||
self.out_indices = out_indices
|
||||
if out_indices is not None:
|
||||
self.out_channels = [self.cfgs[ids - 1][2] for ids in out_indices]
|
||||
else:
|
||||
self.out_channels = self.cfgs[-1][2]
|
||||
|
||||
def forward(self, x):
|
||||
if self.out_indices is not None:
|
||||
return self.forward_det(x)
|
||||
return self.forward_rec(x)
|
||||
|
||||
def forward_det(self, x):
|
||||
outs = []
|
||||
for i, f in enumerate(self.features):
|
||||
x = f(x)
|
||||
if i in self.out_indices:
|
||||
outs.append(x)
|
||||
return outs
|
||||
|
||||
def forward_rec(self, x):
|
||||
for f in self.features:
|
||||
x = f(x)
|
||||
h = x.shape[2]
|
||||
x = nn.functional.avg_pool2d(x, [h, 2])
|
||||
return x
|
||||
|
||||
|
||||
def RepSVTR(in_channels=3):
|
||||
"""
|
||||
Constructs a MobileNetV3-Large model
|
||||
"""
|
||||
# k, t, c, SE, HS, s
|
||||
cfgs = [
|
||||
[3, 2, 96, 1, 0, 1],
|
||||
[3, 2, 96, 0, 0, 1],
|
||||
[3, 2, 96, 0, 0, 1],
|
||||
[3, 2, 192, 0, 1, (2, 1)],
|
||||
[3, 2, 192, 1, 1, 1],
|
||||
[3, 2, 192, 0, 1, 1],
|
||||
[3, 2, 192, 1, 1, 1],
|
||||
[3, 2, 192, 0, 1, 1],
|
||||
[3, 2, 192, 1, 1, 1],
|
||||
[3, 2, 192, 0, 1, 1],
|
||||
[3, 2, 384, 0, 1, (2, 1)],
|
||||
[3, 2, 384, 1, 1, 1],
|
||||
[3, 2, 384, 0, 1, 1],
|
||||
]
|
||||
return RepViT(cfgs, in_channels=in_channels)
|
||||
|
||||
|
||||
def RepSVTR_det(in_channels=3, out_indices=[2, 5, 10, 13]):
|
||||
"""
|
||||
Constructs a MobileNetV3-Large model
|
||||
"""
|
||||
# k, t, c, SE, HS, s
|
||||
cfgs = [
|
||||
[3, 2, 48, 1, 0, 1],
|
||||
[3, 2, 48, 0, 0, 1],
|
||||
[3, 2, 96, 0, 0, 2],
|
||||
[3, 2, 96, 1, 0, 1],
|
||||
[3, 2, 96, 0, 0, 1],
|
||||
[3, 2, 192, 0, 1, 2],
|
||||
[3, 2, 192, 1, 1, 1],
|
||||
[3, 2, 192, 0, 1, 1],
|
||||
[3, 2, 192, 1, 1, 1],
|
||||
[3, 2, 192, 0, 1, 1],
|
||||
[3, 2, 384, 0, 1, 2],
|
||||
[3, 2, 384, 1, 1, 1],
|
||||
[3, 2, 384, 0, 1, 1],
|
||||
]
|
||||
return RepViT(cfgs, in_channels=in_channels, out_indices=out_indices)
|
||||
318
ppocr/modeling/backbones/rec_resnet_31.py
Normal file
318
ppocr/modeling/backbones/rec_resnet_31.py
Normal file
@@ -0,0 +1,318 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/open-mmlab/mmocr/blob/main/mmocr/models/textrecog/layers/conv_layer.py
|
||||
https://github.com/open-mmlab/mmocr/blob/main/mmocr/models/textrecog/backbones/resnet31_ocr.py
|
||||
"""
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import ParamAttr
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
import numpy as np
|
||||
|
||||
__all__ = ["ResNet31"]
|
||||
|
||||
|
||||
def conv3x3(in_channel, out_channel, stride=1, conv_weight_attr=None):
|
||||
return nn.Conv2D(
|
||||
in_channel,
|
||||
out_channel,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
padding=1,
|
||||
weight_attr=conv_weight_attr,
|
||||
bias_attr=False,
|
||||
)
|
||||
|
||||
|
||||
class BasicBlock(nn.Layer):
|
||||
expansion = 1
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
channels,
|
||||
stride=1,
|
||||
downsample=False,
|
||||
conv_weight_attr=None,
|
||||
bn_weight_attr=None,
|
||||
):
|
||||
super().__init__()
|
||||
self.conv1 = conv3x3(
|
||||
in_channels, channels, stride, conv_weight_attr=conv_weight_attr
|
||||
)
|
||||
self.bn1 = nn.BatchNorm2D(channels, weight_attr=bn_weight_attr)
|
||||
self.relu = nn.ReLU()
|
||||
self.conv2 = conv3x3(channels, channels, conv_weight_attr=conv_weight_attr)
|
||||
self.bn2 = nn.BatchNorm2D(channels, weight_attr=bn_weight_attr)
|
||||
self.downsample = downsample
|
||||
if downsample:
|
||||
self.downsample = nn.Sequential(
|
||||
nn.Conv2D(
|
||||
in_channels,
|
||||
channels * self.expansion,
|
||||
1,
|
||||
stride,
|
||||
weight_attr=conv_weight_attr,
|
||||
bias_attr=False,
|
||||
),
|
||||
nn.BatchNorm2D(channels * self.expansion, weight_attr=bn_weight_attr),
|
||||
)
|
||||
else:
|
||||
self.downsample = nn.Sequential()
|
||||
self.stride = stride
|
||||
|
||||
def forward(self, x):
|
||||
residual = x
|
||||
|
||||
out = self.conv1(x)
|
||||
out = self.bn1(out)
|
||||
out = self.relu(out)
|
||||
|
||||
out = self.conv2(out)
|
||||
out = self.bn2(out)
|
||||
|
||||
if self.downsample:
|
||||
residual = self.downsample(x)
|
||||
|
||||
out += residual
|
||||
out = self.relu(out)
|
||||
|
||||
return out
|
||||
|
||||
|
||||
class ResNet31(nn.Layer):
|
||||
"""
|
||||
Args:
|
||||
in_channels (int): Number of channels of input image tensor.
|
||||
layers (list[int]): List of BasicBlock number for each stage.
|
||||
channels (list[int]): List of out_channels of Conv2d layer.
|
||||
out_indices (None | Sequence[int]): Indices of output stages.
|
||||
last_stage_pool (bool): If True, add `MaxPool2d` layer to last stage.
|
||||
init_type (None | str): the config to control the initialization.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
in_channels=3,
|
||||
layers=[1, 2, 5, 3],
|
||||
channels=[64, 128, 256, 256, 512, 512, 512],
|
||||
out_indices=None,
|
||||
last_stage_pool=False,
|
||||
init_type=None,
|
||||
):
|
||||
super(ResNet31, self).__init__()
|
||||
assert isinstance(in_channels, int)
|
||||
assert isinstance(last_stage_pool, bool)
|
||||
|
||||
self.out_indices = out_indices
|
||||
self.last_stage_pool = last_stage_pool
|
||||
|
||||
conv_weight_attr = None
|
||||
bn_weight_attr = None
|
||||
|
||||
if init_type is not None:
|
||||
support_dict = ["KaimingNormal"]
|
||||
assert init_type in support_dict, Exception(
|
||||
"resnet31 only support {}".format(support_dict)
|
||||
)
|
||||
conv_weight_attr = nn.initializer.KaimingNormal()
|
||||
bn_weight_attr = ParamAttr(
|
||||
initializer=nn.initializer.Uniform(), learning_rate=1
|
||||
)
|
||||
|
||||
# conv 1 (Conv Conv)
|
||||
self.conv1_1 = nn.Conv2D(
|
||||
in_channels,
|
||||
channels[0],
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
weight_attr=conv_weight_attr,
|
||||
)
|
||||
self.bn1_1 = nn.BatchNorm2D(channels[0], weight_attr=bn_weight_attr)
|
||||
self.relu1_1 = nn.ReLU()
|
||||
|
||||
self.conv1_2 = nn.Conv2D(
|
||||
channels[0],
|
||||
channels[1],
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
weight_attr=conv_weight_attr,
|
||||
)
|
||||
self.bn1_2 = nn.BatchNorm2D(channels[1], weight_attr=bn_weight_attr)
|
||||
self.relu1_2 = nn.ReLU()
|
||||
|
||||
# conv 2 (Max-pooling, Residual block, Conv)
|
||||
self.pool2 = nn.MaxPool2D(kernel_size=2, stride=2, padding=0, ceil_mode=True)
|
||||
self.block2 = self._make_layer(
|
||||
channels[1],
|
||||
channels[2],
|
||||
layers[0],
|
||||
conv_weight_attr=conv_weight_attr,
|
||||
bn_weight_attr=bn_weight_attr,
|
||||
)
|
||||
self.conv2 = nn.Conv2D(
|
||||
channels[2],
|
||||
channels[2],
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
weight_attr=conv_weight_attr,
|
||||
)
|
||||
self.bn2 = nn.BatchNorm2D(channels[2], weight_attr=bn_weight_attr)
|
||||
self.relu2 = nn.ReLU()
|
||||
|
||||
# conv 3 (Max-pooling, Residual block, Conv)
|
||||
self.pool3 = nn.MaxPool2D(kernel_size=2, stride=2, padding=0, ceil_mode=True)
|
||||
self.block3 = self._make_layer(
|
||||
channels[2],
|
||||
channels[3],
|
||||
layers[1],
|
||||
conv_weight_attr=conv_weight_attr,
|
||||
bn_weight_attr=bn_weight_attr,
|
||||
)
|
||||
self.conv3 = nn.Conv2D(
|
||||
channels[3],
|
||||
channels[3],
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
weight_attr=conv_weight_attr,
|
||||
)
|
||||
self.bn3 = nn.BatchNorm2D(channels[3], weight_attr=bn_weight_attr)
|
||||
self.relu3 = nn.ReLU()
|
||||
|
||||
# conv 4 (Max-pooling, Residual block, Conv)
|
||||
self.pool4 = nn.MaxPool2D(
|
||||
kernel_size=(2, 1), stride=(2, 1), padding=0, ceil_mode=True
|
||||
)
|
||||
self.block4 = self._make_layer(
|
||||
channels[3],
|
||||
channels[4],
|
||||
layers[2],
|
||||
conv_weight_attr=conv_weight_attr,
|
||||
bn_weight_attr=bn_weight_attr,
|
||||
)
|
||||
self.conv4 = nn.Conv2D(
|
||||
channels[4],
|
||||
channels[4],
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
weight_attr=conv_weight_attr,
|
||||
)
|
||||
self.bn4 = nn.BatchNorm2D(channels[4], weight_attr=bn_weight_attr)
|
||||
self.relu4 = nn.ReLU()
|
||||
|
||||
# conv 5 ((Max-pooling), Residual block, Conv)
|
||||
self.pool5 = None
|
||||
if self.last_stage_pool:
|
||||
self.pool5 = nn.MaxPool2D(
|
||||
kernel_size=2, stride=2, padding=0, ceil_mode=True
|
||||
)
|
||||
self.block5 = self._make_layer(
|
||||
channels[4],
|
||||
channels[5],
|
||||
layers[3],
|
||||
conv_weight_attr=conv_weight_attr,
|
||||
bn_weight_attr=bn_weight_attr,
|
||||
)
|
||||
self.conv5 = nn.Conv2D(
|
||||
channels[5],
|
||||
channels[5],
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
weight_attr=conv_weight_attr,
|
||||
)
|
||||
self.bn5 = nn.BatchNorm2D(channels[5], weight_attr=bn_weight_attr)
|
||||
self.relu5 = nn.ReLU()
|
||||
|
||||
self.out_channels = channels[-1]
|
||||
|
||||
def _make_layer(
|
||||
self,
|
||||
input_channels,
|
||||
output_channels,
|
||||
blocks,
|
||||
conv_weight_attr=None,
|
||||
bn_weight_attr=None,
|
||||
):
|
||||
layers = []
|
||||
for _ in range(blocks):
|
||||
downsample = None
|
||||
if input_channels != output_channels:
|
||||
downsample = nn.Sequential(
|
||||
nn.Conv2D(
|
||||
input_channels,
|
||||
output_channels,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
weight_attr=conv_weight_attr,
|
||||
bias_attr=False,
|
||||
),
|
||||
nn.BatchNorm2D(output_channels, weight_attr=bn_weight_attr),
|
||||
)
|
||||
|
||||
layers.append(
|
||||
BasicBlock(
|
||||
input_channels,
|
||||
output_channels,
|
||||
downsample=downsample,
|
||||
conv_weight_attr=conv_weight_attr,
|
||||
bn_weight_attr=bn_weight_attr,
|
||||
)
|
||||
)
|
||||
input_channels = output_channels
|
||||
return nn.Sequential(*layers)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.conv1_1(x)
|
||||
x = self.bn1_1(x)
|
||||
x = self.relu1_1(x)
|
||||
|
||||
x = self.conv1_2(x)
|
||||
x = self.bn1_2(x)
|
||||
x = self.relu1_2(x)
|
||||
|
||||
outs = []
|
||||
for i in range(4):
|
||||
layer_index = i + 2
|
||||
pool_layer = getattr(self, f"pool{layer_index}")
|
||||
block_layer = getattr(self, f"block{layer_index}")
|
||||
conv_layer = getattr(self, f"conv{layer_index}")
|
||||
bn_layer = getattr(self, f"bn{layer_index}")
|
||||
relu_layer = getattr(self, f"relu{layer_index}")
|
||||
|
||||
if pool_layer is not None:
|
||||
x = pool_layer(x)
|
||||
x = block_layer(x)
|
||||
x = conv_layer(x)
|
||||
x = bn_layer(x)
|
||||
x = relu_layer(x)
|
||||
|
||||
outs.append(x)
|
||||
|
||||
if self.out_indices is not None:
|
||||
return tuple([outs[i] for i in self.out_indices])
|
||||
|
||||
return x
|
||||
305
ppocr/modeling/backbones/rec_resnet_32.py
Normal file
305
ppocr/modeling/backbones/rec_resnet_32.py
Normal file
@@ -0,0 +1,305 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/hikopensource/DAVAR-Lab-OCR/davarocr/davar_rcg/models/backbones/ResNet32.py
|
||||
"""
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle.nn as nn
|
||||
|
||||
__all__ = ["ResNet32"]
|
||||
|
||||
conv_weight_attr = nn.initializer.KaimingNormal()
|
||||
|
||||
|
||||
class ResNet32(nn.Layer):
|
||||
"""
|
||||
Feature Extractor is proposed in FAN Ref [1]
|
||||
|
||||
Ref [1]: Focusing Attention: Towards Accurate Text Recognition in Neural Images ICCV-2017
|
||||
"""
|
||||
|
||||
def __init__(self, in_channels, out_channels=512):
|
||||
"""
|
||||
|
||||
Args:
|
||||
in_channels (int): input channel
|
||||
output_channel (int): output channel
|
||||
"""
|
||||
super(ResNet32, self).__init__()
|
||||
self.out_channels = out_channels
|
||||
self.ConvNet = ResNet(in_channels, out_channels, BasicBlock, [1, 2, 5, 3])
|
||||
|
||||
def forward(self, inputs):
|
||||
"""
|
||||
Args:
|
||||
inputs: input feature
|
||||
|
||||
Returns:
|
||||
output feature
|
||||
|
||||
"""
|
||||
return self.ConvNet(inputs)
|
||||
|
||||
|
||||
class BasicBlock(nn.Layer):
|
||||
"""Res-net Basic Block"""
|
||||
|
||||
expansion = 1
|
||||
|
||||
def __init__(
|
||||
self, inplanes, planes, stride=1, downsample=None, norm_type="BN", **kwargs
|
||||
):
|
||||
"""
|
||||
Args:
|
||||
inplanes (int): input channel
|
||||
planes (int): channels of the middle feature
|
||||
stride (int): stride of the convolution
|
||||
downsample (int): type of the down_sample
|
||||
norm_type (str): type of the normalization
|
||||
**kwargs (None): backup parameter
|
||||
"""
|
||||
super(BasicBlock, self).__init__()
|
||||
self.conv1 = self._conv3x3(inplanes, planes)
|
||||
self.bn1 = nn.BatchNorm2D(planes)
|
||||
self.conv2 = self._conv3x3(planes, planes)
|
||||
self.bn2 = nn.BatchNorm2D(planes)
|
||||
self.relu = nn.ReLU()
|
||||
self.downsample = downsample
|
||||
self.stride = stride
|
||||
|
||||
def _conv3x3(self, in_planes, out_planes, stride=1):
|
||||
"""
|
||||
|
||||
Args:
|
||||
in_planes (int): input channel
|
||||
out_planes (int): channels of the middle feature
|
||||
stride (int): stride of the convolution
|
||||
Returns:
|
||||
nn.Layer: Conv2D with kernel = 3
|
||||
|
||||
"""
|
||||
|
||||
return nn.Conv2D(
|
||||
in_planes,
|
||||
out_planes,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
padding=1,
|
||||
weight_attr=conv_weight_attr,
|
||||
bias_attr=False,
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
residual = x
|
||||
|
||||
out = self.conv1(x)
|
||||
out = self.bn1(out)
|
||||
out = self.relu(out)
|
||||
|
||||
out = self.conv2(out)
|
||||
out = self.bn2(out)
|
||||
|
||||
if self.downsample is not None:
|
||||
residual = self.downsample(x)
|
||||
out += residual
|
||||
out = self.relu(out)
|
||||
|
||||
return out
|
||||
|
||||
|
||||
class ResNet(nn.Layer):
|
||||
"""Res-Net network structure"""
|
||||
|
||||
def __init__(self, input_channel, output_channel, block, layers):
|
||||
"""
|
||||
|
||||
Args:
|
||||
input_channel (int): input channel
|
||||
output_channel (int): output channel
|
||||
block (BasicBlock): convolution block
|
||||
layers (list): layers of the block
|
||||
"""
|
||||
super(ResNet, self).__init__()
|
||||
|
||||
self.output_channel_block = [
|
||||
int(output_channel / 4),
|
||||
int(output_channel / 2),
|
||||
output_channel,
|
||||
output_channel,
|
||||
]
|
||||
|
||||
self.inplanes = int(output_channel / 8)
|
||||
self.conv0_1 = nn.Conv2D(
|
||||
input_channel,
|
||||
int(output_channel / 16),
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
weight_attr=conv_weight_attr,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn0_1 = nn.BatchNorm2D(int(output_channel / 16))
|
||||
self.conv0_2 = nn.Conv2D(
|
||||
int(output_channel / 16),
|
||||
self.inplanes,
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
weight_attr=conv_weight_attr,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn0_2 = nn.BatchNorm2D(self.inplanes)
|
||||
self.relu = nn.ReLU()
|
||||
|
||||
self.maxpool1 = nn.MaxPool2D(kernel_size=2, stride=2, padding=0)
|
||||
self.layer1 = self._make_layer(block, self.output_channel_block[0], layers[0])
|
||||
self.conv1 = nn.Conv2D(
|
||||
self.output_channel_block[0],
|
||||
self.output_channel_block[0],
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
weight_attr=conv_weight_attr,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn1 = nn.BatchNorm2D(self.output_channel_block[0])
|
||||
|
||||
self.maxpool2 = nn.MaxPool2D(kernel_size=2, stride=2, padding=0)
|
||||
self.layer2 = self._make_layer(
|
||||
block, self.output_channel_block[1], layers[1], stride=1
|
||||
)
|
||||
self.conv2 = nn.Conv2D(
|
||||
self.output_channel_block[1],
|
||||
self.output_channel_block[1],
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
weight_attr=conv_weight_attr,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn2 = nn.BatchNorm2D(self.output_channel_block[1])
|
||||
|
||||
self.maxpool3 = nn.MaxPool2D(kernel_size=2, stride=(2, 1), padding=(0, 1))
|
||||
self.layer3 = self._make_layer(
|
||||
block, self.output_channel_block[2], layers[2], stride=1
|
||||
)
|
||||
self.conv3 = nn.Conv2D(
|
||||
self.output_channel_block[2],
|
||||
self.output_channel_block[2],
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
weight_attr=conv_weight_attr,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn3 = nn.BatchNorm2D(self.output_channel_block[2])
|
||||
|
||||
self.layer4 = self._make_layer(
|
||||
block, self.output_channel_block[3], layers[3], stride=1
|
||||
)
|
||||
self.conv4_1 = nn.Conv2D(
|
||||
self.output_channel_block[3],
|
||||
self.output_channel_block[3],
|
||||
kernel_size=2,
|
||||
stride=(2, 1),
|
||||
padding=(0, 1),
|
||||
weight_attr=conv_weight_attr,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn4_1 = nn.BatchNorm2D(self.output_channel_block[3])
|
||||
self.conv4_2 = nn.Conv2D(
|
||||
self.output_channel_block[3],
|
||||
self.output_channel_block[3],
|
||||
kernel_size=2,
|
||||
stride=1,
|
||||
padding=0,
|
||||
weight_attr=conv_weight_attr,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn4_2 = nn.BatchNorm2D(self.output_channel_block[3])
|
||||
|
||||
def _make_layer(self, block, planes, blocks, stride=1):
|
||||
"""
|
||||
|
||||
Args:
|
||||
block (block): convolution block
|
||||
planes (int): input channels
|
||||
blocks (list): layers of the block
|
||||
stride (int): stride of the convolution
|
||||
|
||||
Returns:
|
||||
nn.Sequential: the combination of the convolution block
|
||||
|
||||
"""
|
||||
downsample = None
|
||||
if stride != 1 or self.inplanes != planes * block.expansion:
|
||||
downsample = nn.Sequential(
|
||||
nn.Conv2D(
|
||||
self.inplanes,
|
||||
planes * block.expansion,
|
||||
kernel_size=1,
|
||||
stride=stride,
|
||||
weight_attr=conv_weight_attr,
|
||||
bias_attr=False,
|
||||
),
|
||||
nn.BatchNorm2D(planes * block.expansion),
|
||||
)
|
||||
|
||||
layers = list()
|
||||
layers.append(block(self.inplanes, planes, stride, downsample))
|
||||
self.inplanes = planes * block.expansion
|
||||
for _ in range(1, blocks):
|
||||
layers.append(block(self.inplanes, planes))
|
||||
|
||||
return nn.Sequential(*layers)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.conv0_1(x)
|
||||
x = self.bn0_1(x)
|
||||
x = self.relu(x)
|
||||
x = self.conv0_2(x)
|
||||
x = self.bn0_2(x)
|
||||
x = self.relu(x)
|
||||
|
||||
x = self.maxpool1(x)
|
||||
x = self.layer1(x)
|
||||
x = self.conv1(x)
|
||||
x = self.bn1(x)
|
||||
x = self.relu(x)
|
||||
|
||||
x = self.maxpool2(x)
|
||||
x = self.layer2(x)
|
||||
x = self.conv2(x)
|
||||
x = self.bn2(x)
|
||||
x = self.relu(x)
|
||||
|
||||
x = self.maxpool3(x)
|
||||
x = self.layer3(x)
|
||||
x = self.conv3(x)
|
||||
x = self.bn3(x)
|
||||
x = self.relu(x)
|
||||
|
||||
x = self.layer4(x)
|
||||
x = self.conv4_1(x)
|
||||
x = self.bn4_1(x)
|
||||
x = self.relu(x)
|
||||
x = self.conv4_2(x)
|
||||
x = self.bn4_2(x)
|
||||
x = self.relu(x)
|
||||
return x
|
||||
150
ppocr/modeling/backbones/rec_resnet_45.py
Normal file
150
ppocr/modeling/backbones/rec_resnet_45.py
Normal file
@@ -0,0 +1,150 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/FangShancheng/ABINet/tree/main/modules
|
||||
"""
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import ParamAttr
|
||||
from paddle.nn.initializer import KaimingNormal
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
import numpy as np
|
||||
import math
|
||||
|
||||
__all__ = ["ResNet45"]
|
||||
|
||||
|
||||
def conv1x1(in_planes, out_planes, stride=1):
|
||||
return nn.Conv2D(
|
||||
in_planes,
|
||||
out_planes,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
weight_attr=ParamAttr(initializer=KaimingNormal()),
|
||||
bias_attr=False,
|
||||
)
|
||||
|
||||
|
||||
def conv3x3(in_channel, out_channel, stride=1):
|
||||
return nn.Conv2D(
|
||||
in_channel,
|
||||
out_channel,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
padding=1,
|
||||
weight_attr=ParamAttr(initializer=KaimingNormal()),
|
||||
bias_attr=False,
|
||||
)
|
||||
|
||||
|
||||
class BasicBlock(nn.Layer):
|
||||
expansion = 1
|
||||
|
||||
def __init__(self, in_channels, channels, stride=1, downsample=None):
|
||||
super().__init__()
|
||||
self.conv1 = conv1x1(in_channels, channels)
|
||||
self.bn1 = nn.BatchNorm2D(channels)
|
||||
self.relu = nn.ReLU()
|
||||
self.conv2 = conv3x3(channels, channels, stride)
|
||||
self.bn2 = nn.BatchNorm2D(channels)
|
||||
self.downsample = downsample
|
||||
self.stride = stride
|
||||
|
||||
def forward(self, x):
|
||||
residual = x
|
||||
|
||||
out = self.conv1(x)
|
||||
out = self.bn1(out)
|
||||
out = self.relu(out)
|
||||
|
||||
out = self.conv2(out)
|
||||
out = self.bn2(out)
|
||||
|
||||
if self.downsample is not None:
|
||||
residual = self.downsample(x)
|
||||
out += residual
|
||||
out = self.relu(out)
|
||||
|
||||
return out
|
||||
|
||||
|
||||
class ResNet45(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels=3,
|
||||
block=BasicBlock,
|
||||
layers=[3, 4, 6, 6, 3],
|
||||
strides=[2, 1, 2, 1, 1],
|
||||
):
|
||||
self.inplanes = 32
|
||||
super(ResNet45, self).__init__()
|
||||
self.conv1 = nn.Conv2D(
|
||||
in_channels,
|
||||
32,
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
weight_attr=ParamAttr(initializer=KaimingNormal()),
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn1 = nn.BatchNorm2D(32)
|
||||
self.relu = nn.ReLU()
|
||||
|
||||
self.layer1 = self._make_layer(block, 32, layers[0], stride=strides[0])
|
||||
self.layer2 = self._make_layer(block, 64, layers[1], stride=strides[1])
|
||||
self.layer3 = self._make_layer(block, 128, layers[2], stride=strides[2])
|
||||
self.layer4 = self._make_layer(block, 256, layers[3], stride=strides[3])
|
||||
self.layer5 = self._make_layer(block, 512, layers[4], stride=strides[4])
|
||||
self.out_channels = 512
|
||||
|
||||
def _make_layer(self, block, planes, blocks, stride=1):
|
||||
downsample = None
|
||||
if stride != 1 or self.inplanes != planes * block.expansion:
|
||||
# downsample = True
|
||||
downsample = nn.Sequential(
|
||||
nn.Conv2D(
|
||||
self.inplanes,
|
||||
planes * block.expansion,
|
||||
kernel_size=1,
|
||||
stride=stride,
|
||||
weight_attr=ParamAttr(initializer=KaimingNormal()),
|
||||
bias_attr=False,
|
||||
),
|
||||
nn.BatchNorm2D(planes * block.expansion),
|
||||
)
|
||||
|
||||
layers = []
|
||||
layers.append(block(self.inplanes, planes, stride, downsample))
|
||||
self.inplanes = planes * block.expansion
|
||||
for i in range(1, blocks):
|
||||
layers.append(block(self.inplanes, planes))
|
||||
|
||||
return nn.Sequential(*layers)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.conv1(x)
|
||||
x = self.bn1(x)
|
||||
x = self.relu(x)
|
||||
x = self.layer1(x)
|
||||
x = self.layer2(x)
|
||||
x = self.layer3(x)
|
||||
x = self.layer4(x)
|
||||
x = self.layer5(x)
|
||||
return x
|
||||
141
ppocr/modeling/backbones/rec_resnet_aster.py
Normal file
141
ppocr/modeling/backbones/rec_resnet_aster.py
Normal file
@@ -0,0 +1,141 @@
|
||||
# copyright (c) 2020 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/ayumiymk/aster.pytorch/blob/master/lib/models/resnet_aster.py
|
||||
"""
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
|
||||
import sys
|
||||
import math
|
||||
|
||||
|
||||
def conv3x3(in_planes, out_planes, stride=1):
|
||||
"""3x3 convolution with padding"""
|
||||
return nn.Conv2D(
|
||||
in_planes, out_planes, kernel_size=3, stride=stride, padding=1, bias_attr=False
|
||||
)
|
||||
|
||||
|
||||
def conv1x1(in_planes, out_planes, stride=1):
|
||||
"""1x1 convolution"""
|
||||
return nn.Conv2D(
|
||||
in_planes, out_planes, kernel_size=1, stride=stride, bias_attr=False
|
||||
)
|
||||
|
||||
|
||||
def get_sinusoid_encoding(n_position, feat_dim, wave_length=10000):
|
||||
# [n_position]
|
||||
positions = paddle.arange(0, n_position)
|
||||
# [feat_dim]
|
||||
dim_range = paddle.arange(0, feat_dim)
|
||||
dim_range = paddle.pow(wave_length, 2 * (dim_range // 2) / feat_dim)
|
||||
# [n_position, feat_dim]
|
||||
angles = paddle.unsqueeze(positions, axis=1) / paddle.unsqueeze(dim_range, axis=0)
|
||||
angles = paddle.cast(angles, "float32")
|
||||
angles[:, 0::2] = paddle.sin(angles[:, 0::2])
|
||||
angles[:, 1::2] = paddle.cos(angles[:, 1::2])
|
||||
return angles
|
||||
|
||||
|
||||
class AsterBlock(nn.Layer):
|
||||
def __init__(self, inplanes, planes, stride=1, downsample=None):
|
||||
super(AsterBlock, self).__init__()
|
||||
self.conv1 = conv1x1(inplanes, planes, stride)
|
||||
self.bn1 = nn.BatchNorm2D(planes)
|
||||
self.relu = nn.ReLU()
|
||||
self.conv2 = conv3x3(planes, planes)
|
||||
self.bn2 = nn.BatchNorm2D(planes)
|
||||
self.downsample = downsample
|
||||
self.stride = stride
|
||||
|
||||
def forward(self, x):
|
||||
residual = x
|
||||
out = self.conv1(x)
|
||||
out = self.bn1(out)
|
||||
out = self.relu(out)
|
||||
out = self.conv2(out)
|
||||
out = self.bn2(out)
|
||||
|
||||
if self.downsample is not None:
|
||||
residual = self.downsample(x)
|
||||
out += residual
|
||||
out = self.relu(out)
|
||||
return out
|
||||
|
||||
|
||||
class ResNet_ASTER(nn.Layer):
|
||||
"""For aster or crnn"""
|
||||
|
||||
def __init__(self, with_lstm=True, n_group=1, in_channels=3):
|
||||
super(ResNet_ASTER, self).__init__()
|
||||
self.with_lstm = with_lstm
|
||||
self.n_group = n_group
|
||||
|
||||
self.layer0 = nn.Sequential(
|
||||
nn.Conv2D(
|
||||
in_channels,
|
||||
32,
|
||||
kernel_size=(3, 3),
|
||||
stride=1,
|
||||
padding=1,
|
||||
bias_attr=False,
|
||||
),
|
||||
nn.BatchNorm2D(32),
|
||||
nn.ReLU(),
|
||||
)
|
||||
|
||||
self.inplanes = 32
|
||||
self.layer1 = self._make_layer(32, 3, [2, 2]) # [16, 50]
|
||||
self.layer2 = self._make_layer(64, 4, [2, 2]) # [8, 25]
|
||||
self.layer3 = self._make_layer(128, 6, [2, 1]) # [4, 25]
|
||||
self.layer4 = self._make_layer(256, 6, [2, 1]) # [2, 25]
|
||||
self.layer5 = self._make_layer(512, 3, [2, 1]) # [1, 25]
|
||||
|
||||
if with_lstm:
|
||||
self.rnn = nn.LSTM(512, 256, direction="bidirect", num_layers=2)
|
||||
self.out_channels = 2 * 256
|
||||
else:
|
||||
self.out_channels = 512
|
||||
|
||||
def _make_layer(self, planes, blocks, stride):
|
||||
downsample = None
|
||||
if stride != [1, 1] or self.inplanes != planes:
|
||||
downsample = nn.Sequential(
|
||||
conv1x1(self.inplanes, planes, stride), nn.BatchNorm2D(planes)
|
||||
)
|
||||
|
||||
layers = []
|
||||
layers.append(AsterBlock(self.inplanes, planes, stride, downsample))
|
||||
self.inplanes = planes
|
||||
for _ in range(1, blocks):
|
||||
layers.append(AsterBlock(self.inplanes, planes))
|
||||
return nn.Sequential(*layers)
|
||||
|
||||
def forward(self, x):
|
||||
x0 = self.layer0(x)
|
||||
x1 = self.layer1(x0)
|
||||
x2 = self.layer2(x1)
|
||||
x3 = self.layer3(x2)
|
||||
x4 = self.layer4(x3)
|
||||
x5 = self.layer5(x4)
|
||||
|
||||
cnn_feat = x5.squeeze(2) # [N, c, w]
|
||||
cnn_feat = paddle.transpose(cnn_feat, perm=[0, 2, 1])
|
||||
if self.with_lstm:
|
||||
rnn_feat, _ = self.rnn(cnn_feat)
|
||||
return rnn_feat
|
||||
else:
|
||||
return cnn_feat
|
||||
317
ppocr/modeling/backbones/rec_resnet_fpn.py
Normal file
317
ppocr/modeling/backbones/rec_resnet_fpn.py
Normal file
@@ -0,0 +1,317 @@
|
||||
# copyright (c) 2020 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
from paddle import nn, ParamAttr
|
||||
from paddle.nn import functional as F
|
||||
import paddle
|
||||
import numpy as np
|
||||
|
||||
__all__ = ["ResNetFPN"]
|
||||
|
||||
|
||||
class ResNetFPN(nn.Layer):
|
||||
def __init__(self, in_channels=1, layers=50, **kwargs):
|
||||
super(ResNetFPN, self).__init__()
|
||||
supported_layers = {
|
||||
18: {"depth": [2, 2, 2, 2], "block_class": BasicBlock},
|
||||
34: {"depth": [3, 4, 6, 3], "block_class": BasicBlock},
|
||||
50: {"depth": [3, 4, 6, 3], "block_class": BottleneckBlock},
|
||||
101: {"depth": [3, 4, 23, 3], "block_class": BottleneckBlock},
|
||||
152: {"depth": [3, 8, 36, 3], "block_class": BottleneckBlock},
|
||||
}
|
||||
stride_list = [(2, 2), (2, 2), (1, 1), (1, 1)]
|
||||
num_filters = [64, 128, 256, 512]
|
||||
self.depth = supported_layers[layers]["depth"]
|
||||
self.F = []
|
||||
self.conv = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=64,
|
||||
kernel_size=7,
|
||||
stride=2,
|
||||
act="relu",
|
||||
name="conv1",
|
||||
)
|
||||
self.block_list = []
|
||||
in_ch = 64
|
||||
if layers >= 50:
|
||||
for block in range(len(self.depth)):
|
||||
for i in range(self.depth[block]):
|
||||
if layers in [101, 152] and block == 2:
|
||||
if i == 0:
|
||||
conv_name = "res" + str(block + 2) + "a"
|
||||
else:
|
||||
conv_name = "res" + str(block + 2) + "b" + str(i)
|
||||
else:
|
||||
conv_name = "res" + str(block + 2) + chr(97 + i)
|
||||
block_list = self.add_sublayer(
|
||||
"bottleneckBlock_{}_{}".format(block, i),
|
||||
BottleneckBlock(
|
||||
in_channels=in_ch,
|
||||
out_channels=num_filters[block],
|
||||
stride=stride_list[block] if i == 0 else 1,
|
||||
name=conv_name,
|
||||
),
|
||||
)
|
||||
in_ch = num_filters[block] * 4
|
||||
self.block_list.append(block_list)
|
||||
self.F.append(block_list)
|
||||
else:
|
||||
for block in range(len(self.depth)):
|
||||
for i in range(self.depth[block]):
|
||||
conv_name = "res" + str(block + 2) + chr(97 + i)
|
||||
if i == 0 and block != 0:
|
||||
stride = (2, 1)
|
||||
else:
|
||||
stride = (1, 1)
|
||||
basic_block = self.add_sublayer(
|
||||
conv_name,
|
||||
BasicBlock(
|
||||
in_channels=in_ch,
|
||||
out_channels=num_filters[block],
|
||||
stride=stride_list[block] if i == 0 else 1,
|
||||
is_first=block == i == 0,
|
||||
name=conv_name,
|
||||
),
|
||||
)
|
||||
in_ch = basic_block.out_channels
|
||||
self.block_list.append(basic_block)
|
||||
out_ch_list = [in_ch // 4, in_ch // 2, in_ch]
|
||||
self.base_block = []
|
||||
self.conv_trans = []
|
||||
self.bn_block = []
|
||||
for i in [-2, -3]:
|
||||
in_channels = out_ch_list[i + 1] + out_ch_list[i]
|
||||
|
||||
self.base_block.append(
|
||||
self.add_sublayer(
|
||||
"F_{}_base_block_0".format(i),
|
||||
nn.Conv2D(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_ch_list[i],
|
||||
kernel_size=1,
|
||||
weight_attr=ParamAttr(trainable=True),
|
||||
bias_attr=ParamAttr(trainable=True),
|
||||
),
|
||||
)
|
||||
)
|
||||
self.base_block.append(
|
||||
self.add_sublayer(
|
||||
"F_{}_base_block_1".format(i),
|
||||
nn.Conv2D(
|
||||
in_channels=out_ch_list[i],
|
||||
out_channels=out_ch_list[i],
|
||||
kernel_size=3,
|
||||
padding=1,
|
||||
weight_attr=ParamAttr(trainable=True),
|
||||
bias_attr=ParamAttr(trainable=True),
|
||||
),
|
||||
)
|
||||
)
|
||||
self.base_block.append(
|
||||
self.add_sublayer(
|
||||
"F_{}_base_block_2".format(i),
|
||||
nn.BatchNorm(
|
||||
num_channels=out_ch_list[i],
|
||||
act="relu",
|
||||
param_attr=ParamAttr(trainable=True),
|
||||
bias_attr=ParamAttr(trainable=True),
|
||||
),
|
||||
)
|
||||
)
|
||||
self.base_block.append(
|
||||
self.add_sublayer(
|
||||
"F_{}_base_block_3".format(i),
|
||||
nn.Conv2D(
|
||||
in_channels=out_ch_list[i],
|
||||
out_channels=512,
|
||||
kernel_size=1,
|
||||
bias_attr=ParamAttr(trainable=True),
|
||||
weight_attr=ParamAttr(trainable=True),
|
||||
),
|
||||
)
|
||||
)
|
||||
self.out_channels = 512
|
||||
|
||||
def __call__(self, x):
|
||||
x = self.conv(x)
|
||||
fpn_list = []
|
||||
F = []
|
||||
for i in range(len(self.depth)):
|
||||
fpn_list.append(np.sum(self.depth[: i + 1]))
|
||||
|
||||
for i, block in enumerate(self.block_list):
|
||||
x = block(x)
|
||||
for number in fpn_list:
|
||||
if i + 1 == number:
|
||||
F.append(x)
|
||||
base = F[-1]
|
||||
|
||||
j = 0
|
||||
for i, block in enumerate(self.base_block):
|
||||
if i % 3 == 0 and i < 6:
|
||||
j = j + 1
|
||||
b, c, w, h = F[-j - 1].shape
|
||||
if [w, h] == list(base.shape[2:]):
|
||||
base = base
|
||||
else:
|
||||
base = self.conv_trans[j - 1](base)
|
||||
base = self.bn_block[j - 1](base)
|
||||
base = paddle.concat([base, F[-j - 1]], axis=1)
|
||||
base = block(base)
|
||||
return base
|
||||
|
||||
|
||||
class ConvBNLayer(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
kernel_size,
|
||||
stride=1,
|
||||
groups=1,
|
||||
act=None,
|
||||
name=None,
|
||||
):
|
||||
super(ConvBNLayer, self).__init__()
|
||||
self.conv = nn.Conv2D(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=2 if stride == (1, 1) else kernel_size,
|
||||
dilation=2 if stride == (1, 1) else 1,
|
||||
stride=stride,
|
||||
padding=(kernel_size - 1) // 2,
|
||||
groups=groups,
|
||||
weight_attr=ParamAttr(name=name + ".conv2d.output.1.w_0"),
|
||||
bias_attr=False,
|
||||
)
|
||||
|
||||
if name == "conv1":
|
||||
bn_name = "bn_" + name
|
||||
else:
|
||||
bn_name = "bn" + name[3:]
|
||||
self.bn = nn.BatchNorm(
|
||||
num_channels=out_channels,
|
||||
act=act,
|
||||
param_attr=ParamAttr(name=name + ".output.1.w_0"),
|
||||
bias_attr=ParamAttr(name=name + ".output.1.b_0"),
|
||||
moving_mean_name=bn_name + "_mean",
|
||||
moving_variance_name=bn_name + "_variance",
|
||||
)
|
||||
|
||||
def __call__(self, x):
|
||||
x = self.conv(x)
|
||||
x = self.bn(x)
|
||||
return x
|
||||
|
||||
|
||||
class ShortCut(nn.Layer):
|
||||
def __init__(self, in_channels, out_channels, stride, name, is_first=False):
|
||||
super(ShortCut, self).__init__()
|
||||
self.use_conv = True
|
||||
|
||||
if in_channels != out_channels or stride != 1 or is_first == True:
|
||||
if stride == (1, 1):
|
||||
self.conv = ConvBNLayer(in_channels, out_channels, 1, 1, name=name)
|
||||
else: # stride==(2,2)
|
||||
self.conv = ConvBNLayer(in_channels, out_channels, 1, stride, name=name)
|
||||
else:
|
||||
self.use_conv = False
|
||||
|
||||
def forward(self, x):
|
||||
if self.use_conv:
|
||||
x = self.conv(x)
|
||||
return x
|
||||
|
||||
|
||||
class BottleneckBlock(nn.Layer):
|
||||
def __init__(self, in_channels, out_channels, stride, name):
|
||||
super(BottleneckBlock, self).__init__()
|
||||
self.conv0 = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=1,
|
||||
act="relu",
|
||||
name=name + "_branch2a",
|
||||
)
|
||||
self.conv1 = ConvBNLayer(
|
||||
in_channels=out_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
act="relu",
|
||||
name=name + "_branch2b",
|
||||
)
|
||||
|
||||
self.conv2 = ConvBNLayer(
|
||||
in_channels=out_channels,
|
||||
out_channels=out_channels * 4,
|
||||
kernel_size=1,
|
||||
act=None,
|
||||
name=name + "_branch2c",
|
||||
)
|
||||
|
||||
self.short = ShortCut(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels * 4,
|
||||
stride=stride,
|
||||
is_first=False,
|
||||
name=name + "_branch1",
|
||||
)
|
||||
self.out_channels = out_channels * 4
|
||||
|
||||
def forward(self, x):
|
||||
y = self.conv0(x)
|
||||
y = self.conv1(y)
|
||||
y = self.conv2(y)
|
||||
y = y + self.short(x)
|
||||
y = F.relu(y)
|
||||
return y
|
||||
|
||||
|
||||
class BasicBlock(nn.Layer):
|
||||
def __init__(self, in_channels, out_channels, stride, name, is_first):
|
||||
super(BasicBlock, self).__init__()
|
||||
self.conv0 = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=3,
|
||||
act="relu",
|
||||
stride=stride,
|
||||
name=name + "_branch2a",
|
||||
)
|
||||
self.conv1 = ConvBNLayer(
|
||||
in_channels=out_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=3,
|
||||
act=None,
|
||||
name=name + "_branch2b",
|
||||
)
|
||||
self.short = ShortCut(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
stride=stride,
|
||||
is_first=is_first,
|
||||
name=name + "_branch1",
|
||||
)
|
||||
self.out_channels = out_channels
|
||||
|
||||
def forward(self, x):
|
||||
y = self.conv0(x)
|
||||
y = self.conv1(y)
|
||||
y = y + self.short(x)
|
||||
return F.relu(y)
|
||||
359
ppocr/modeling/backbones/rec_resnet_rfl.py
Normal file
359
ppocr/modeling/backbones/rec_resnet_rfl.py
Normal file
@@ -0,0 +1,359 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/hikopensource/DAVAR-Lab-OCR/blob/main/davarocr/davar_rcg/models/backbones/ResNetRFL.py
|
||||
"""
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
|
||||
from paddle.nn.initializer import TruncatedNormal, Constant, Normal, KaimingNormal
|
||||
|
||||
kaiming_init_ = KaimingNormal()
|
||||
zeros_ = Constant(value=0.0)
|
||||
ones_ = Constant(value=1.0)
|
||||
|
||||
|
||||
class BasicBlock(nn.Layer):
|
||||
"""Res-net Basic Block"""
|
||||
|
||||
expansion = 1
|
||||
|
||||
def __init__(
|
||||
self, inplanes, planes, stride=1, downsample=None, norm_type="BN", **kwargs
|
||||
):
|
||||
"""
|
||||
Args:
|
||||
inplanes (int): input channel
|
||||
planes (int): channels of the middle feature
|
||||
stride (int): stride of the convolution
|
||||
downsample (int): type of the down_sample
|
||||
norm_type (str): type of the normalization
|
||||
**kwargs (None): backup parameter
|
||||
"""
|
||||
super(BasicBlock, self).__init__()
|
||||
self.conv1 = self._conv3x3(inplanes, planes)
|
||||
self.bn1 = nn.BatchNorm(planes)
|
||||
self.conv2 = self._conv3x3(planes, planes)
|
||||
self.bn2 = nn.BatchNorm(planes)
|
||||
self.relu = nn.ReLU()
|
||||
self.downsample = downsample
|
||||
self.stride = stride
|
||||
|
||||
def _conv3x3(self, in_planes, out_planes, stride=1):
|
||||
return nn.Conv2D(
|
||||
in_planes,
|
||||
out_planes,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
padding=1,
|
||||
bias_attr=False,
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
residual = x
|
||||
|
||||
out = self.conv1(x)
|
||||
out = self.bn1(out)
|
||||
out = self.relu(out)
|
||||
|
||||
out = self.conv2(out)
|
||||
out = self.bn2(out)
|
||||
|
||||
if self.downsample is not None:
|
||||
residual = self.downsample(x)
|
||||
out += residual
|
||||
out = self.relu(out)
|
||||
|
||||
return out
|
||||
|
||||
|
||||
class ResNetRFL(nn.Layer):
|
||||
def __init__(self, in_channels, out_channels=512, use_cnt=True, use_seq=True):
|
||||
"""
|
||||
|
||||
Args:
|
||||
in_channels (int): input channel
|
||||
out_channels (int): output channel
|
||||
"""
|
||||
super(ResNetRFL, self).__init__()
|
||||
assert use_cnt or use_seq
|
||||
self.use_cnt, self.use_seq = use_cnt, use_seq
|
||||
self.backbone = RFLBase(in_channels)
|
||||
|
||||
self.out_channels = out_channels
|
||||
self.out_channels_block = [
|
||||
int(self.out_channels / 4),
|
||||
int(self.out_channels / 2),
|
||||
self.out_channels,
|
||||
self.out_channels,
|
||||
]
|
||||
block = BasicBlock
|
||||
layers = [1, 2, 5, 3]
|
||||
self.inplanes = int(self.out_channels // 2)
|
||||
|
||||
self.relu = nn.ReLU()
|
||||
if self.use_seq:
|
||||
self.maxpool3 = nn.MaxPool2D(kernel_size=2, stride=(2, 1), padding=(0, 1))
|
||||
self.layer3 = self._make_layer(
|
||||
block, self.out_channels_block[2], layers[2], stride=1
|
||||
)
|
||||
self.conv3 = nn.Conv2D(
|
||||
self.out_channels_block[2],
|
||||
self.out_channels_block[2],
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn3 = nn.BatchNorm(self.out_channels_block[2])
|
||||
|
||||
self.layer4 = self._make_layer(
|
||||
block, self.out_channels_block[3], layers[3], stride=1
|
||||
)
|
||||
self.conv4_1 = nn.Conv2D(
|
||||
self.out_channels_block[3],
|
||||
self.out_channels_block[3],
|
||||
kernel_size=2,
|
||||
stride=(2, 1),
|
||||
padding=(0, 1),
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn4_1 = nn.BatchNorm(self.out_channels_block[3])
|
||||
self.conv4_2 = nn.Conv2D(
|
||||
self.out_channels_block[3],
|
||||
self.out_channels_block[3],
|
||||
kernel_size=2,
|
||||
stride=1,
|
||||
padding=0,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn4_2 = nn.BatchNorm(self.out_channels_block[3])
|
||||
|
||||
if self.use_cnt:
|
||||
self.inplanes = int(self.out_channels // 2)
|
||||
self.v_maxpool3 = nn.MaxPool2D(kernel_size=2, stride=(2, 1), padding=(0, 1))
|
||||
self.v_layer3 = self._make_layer(
|
||||
block, self.out_channels_block[2], layers[2], stride=1
|
||||
)
|
||||
self.v_conv3 = nn.Conv2D(
|
||||
self.out_channels_block[2],
|
||||
self.out_channels_block[2],
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.v_bn3 = nn.BatchNorm(self.out_channels_block[2])
|
||||
|
||||
self.v_layer4 = self._make_layer(
|
||||
block, self.out_channels_block[3], layers[3], stride=1
|
||||
)
|
||||
self.v_conv4_1 = nn.Conv2D(
|
||||
self.out_channels_block[3],
|
||||
self.out_channels_block[3],
|
||||
kernel_size=2,
|
||||
stride=(2, 1),
|
||||
padding=(0, 1),
|
||||
bias_attr=False,
|
||||
)
|
||||
self.v_bn4_1 = nn.BatchNorm(self.out_channels_block[3])
|
||||
self.v_conv4_2 = nn.Conv2D(
|
||||
self.out_channels_block[3],
|
||||
self.out_channels_block[3],
|
||||
kernel_size=2,
|
||||
stride=1,
|
||||
padding=0,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.v_bn4_2 = nn.BatchNorm(self.out_channels_block[3])
|
||||
|
||||
def _make_layer(self, block, planes, blocks, stride=1):
|
||||
downsample = None
|
||||
if stride != 1 or self.inplanes != planes * block.expansion:
|
||||
downsample = nn.Sequential(
|
||||
nn.Conv2D(
|
||||
self.inplanes,
|
||||
planes * block.expansion,
|
||||
kernel_size=1,
|
||||
stride=stride,
|
||||
bias_attr=False,
|
||||
),
|
||||
nn.BatchNorm(planes * block.expansion),
|
||||
)
|
||||
|
||||
layers = list()
|
||||
layers.append(block(self.inplanes, planes, stride, downsample))
|
||||
self.inplanes = planes * block.expansion
|
||||
for _ in range(1, blocks):
|
||||
layers.append(block(self.inplanes, planes))
|
||||
|
||||
return nn.Sequential(*layers)
|
||||
|
||||
def forward(self, inputs):
|
||||
x_1 = self.backbone(inputs)
|
||||
|
||||
if self.use_cnt:
|
||||
v_x = self.v_maxpool3(x_1)
|
||||
v_x = self.v_layer3(v_x)
|
||||
v_x = self.v_conv3(v_x)
|
||||
v_x = self.v_bn3(v_x)
|
||||
visual_feature_2 = self.relu(v_x)
|
||||
|
||||
v_x = self.v_layer4(visual_feature_2)
|
||||
v_x = self.v_conv4_1(v_x)
|
||||
v_x = self.v_bn4_1(v_x)
|
||||
v_x = self.relu(v_x)
|
||||
v_x = self.v_conv4_2(v_x)
|
||||
v_x = self.v_bn4_2(v_x)
|
||||
visual_feature_3 = self.relu(v_x)
|
||||
else:
|
||||
visual_feature_3 = None
|
||||
if self.use_seq:
|
||||
x = self.maxpool3(x_1)
|
||||
x = self.layer3(x)
|
||||
x = self.conv3(x)
|
||||
x = self.bn3(x)
|
||||
x_2 = self.relu(x)
|
||||
|
||||
x = self.layer4(x_2)
|
||||
x = self.conv4_1(x)
|
||||
x = self.bn4_1(x)
|
||||
x = self.relu(x)
|
||||
x = self.conv4_2(x)
|
||||
x = self.bn4_2(x)
|
||||
x_3 = self.relu(x)
|
||||
else:
|
||||
x_3 = None
|
||||
|
||||
return [visual_feature_3, x_3]
|
||||
|
||||
|
||||
class ResNetBase(nn.Layer):
|
||||
def __init__(self, in_channels, out_channels, block, layers):
|
||||
super(ResNetBase, self).__init__()
|
||||
|
||||
self.out_channels_block = [
|
||||
int(out_channels / 4),
|
||||
int(out_channels / 2),
|
||||
out_channels,
|
||||
out_channels,
|
||||
]
|
||||
|
||||
self.inplanes = int(out_channels / 8)
|
||||
self.conv0_1 = nn.Conv2D(
|
||||
in_channels,
|
||||
int(out_channels / 16),
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn0_1 = nn.BatchNorm(int(out_channels / 16))
|
||||
self.conv0_2 = nn.Conv2D(
|
||||
int(out_channels / 16),
|
||||
self.inplanes,
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn0_2 = nn.BatchNorm(self.inplanes)
|
||||
self.relu = nn.ReLU()
|
||||
|
||||
self.maxpool1 = nn.MaxPool2D(kernel_size=2, stride=2, padding=0)
|
||||
self.layer1 = self._make_layer(block, self.out_channels_block[0], layers[0])
|
||||
self.conv1 = nn.Conv2D(
|
||||
self.out_channels_block[0],
|
||||
self.out_channels_block[0],
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn1 = nn.BatchNorm(self.out_channels_block[0])
|
||||
|
||||
self.maxpool2 = nn.MaxPool2D(kernel_size=2, stride=2, padding=0)
|
||||
self.layer2 = self._make_layer(
|
||||
block, self.out_channels_block[1], layers[1], stride=1
|
||||
)
|
||||
self.conv2 = nn.Conv2D(
|
||||
self.out_channels_block[1],
|
||||
self.out_channels_block[1],
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=1,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.bn2 = nn.BatchNorm(self.out_channels_block[1])
|
||||
|
||||
def _make_layer(self, block, planes, blocks, stride=1):
|
||||
downsample = None
|
||||
if stride != 1 or self.inplanes != planes * block.expansion:
|
||||
downsample = nn.Sequential(
|
||||
nn.Conv2D(
|
||||
self.inplanes,
|
||||
planes * block.expansion,
|
||||
kernel_size=1,
|
||||
stride=stride,
|
||||
bias_attr=False,
|
||||
),
|
||||
nn.BatchNorm(planes * block.expansion),
|
||||
)
|
||||
|
||||
layers = list()
|
||||
layers.append(block(self.inplanes, planes, stride, downsample))
|
||||
self.inplanes = planes * block.expansion
|
||||
for _ in range(1, blocks):
|
||||
layers.append(block(self.inplanes, planes))
|
||||
|
||||
return nn.Sequential(*layers)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.conv0_1(x)
|
||||
x = self.bn0_1(x)
|
||||
x = self.relu(x)
|
||||
x = self.conv0_2(x)
|
||||
x = self.bn0_2(x)
|
||||
x = self.relu(x)
|
||||
|
||||
x = self.maxpool1(x)
|
||||
x = self.layer1(x)
|
||||
x = self.conv1(x)
|
||||
x = self.bn1(x)
|
||||
x = self.relu(x)
|
||||
|
||||
x = self.maxpool2(x)
|
||||
x = self.layer2(x)
|
||||
x = self.conv2(x)
|
||||
x = self.bn2(x)
|
||||
x = self.relu(x)
|
||||
|
||||
return x
|
||||
|
||||
|
||||
class RFLBase(nn.Layer):
|
||||
"""Reciprocal feature learning share backbone network"""
|
||||
|
||||
def __init__(self, in_channels, out_channels=512):
|
||||
super(RFLBase, self).__init__()
|
||||
self.ConvNet = ResNetBase(in_channels, out_channels, BasicBlock, [1, 2, 5, 3])
|
||||
|
||||
def forward(self, inputs):
|
||||
return self.ConvNet(inputs)
|
||||
313
ppocr/modeling/backbones/rec_resnet_vd.py
Normal file
313
ppocr/modeling/backbones/rec_resnet_vd.py
Normal file
@@ -0,0 +1,313 @@
|
||||
# copyright (c) 2020 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import ParamAttr
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
|
||||
__all__ = ["ResNet"]
|
||||
|
||||
|
||||
class ConvBNLayer(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
kernel_size,
|
||||
stride=1,
|
||||
groups=1,
|
||||
is_vd_mode=False,
|
||||
act=None,
|
||||
name=None,
|
||||
):
|
||||
super(ConvBNLayer, self).__init__()
|
||||
|
||||
self.is_vd_mode = is_vd_mode
|
||||
self._pool2d_avg = nn.AvgPool2D(
|
||||
kernel_size=stride, stride=stride, padding=0, ceil_mode=True
|
||||
)
|
||||
self._conv = nn.Conv2D(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=kernel_size,
|
||||
stride=1 if is_vd_mode else stride,
|
||||
padding=(kernel_size - 1) // 2,
|
||||
groups=groups,
|
||||
weight_attr=ParamAttr(name=name + "_weights"),
|
||||
bias_attr=False,
|
||||
)
|
||||
if name == "conv1":
|
||||
bn_name = "bn_" + name
|
||||
else:
|
||||
bn_name = "bn" + name[3:]
|
||||
self._batch_norm = nn.BatchNorm(
|
||||
out_channels,
|
||||
act=act,
|
||||
param_attr=ParamAttr(name=bn_name + "_scale"),
|
||||
bias_attr=ParamAttr(bn_name + "_offset"),
|
||||
moving_mean_name=bn_name + "_mean",
|
||||
moving_variance_name=bn_name + "_variance",
|
||||
)
|
||||
|
||||
def forward(self, inputs):
|
||||
if self.is_vd_mode:
|
||||
inputs = self._pool2d_avg(inputs)
|
||||
y = self._conv(inputs)
|
||||
y = self._batch_norm(y)
|
||||
return y
|
||||
|
||||
|
||||
class BottleneckBlock(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
stride,
|
||||
shortcut=True,
|
||||
if_first=False,
|
||||
name=None,
|
||||
):
|
||||
super(BottleneckBlock, self).__init__()
|
||||
|
||||
self.conv0 = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=1,
|
||||
act="relu",
|
||||
name=name + "_branch2a",
|
||||
)
|
||||
self.conv1 = ConvBNLayer(
|
||||
in_channels=out_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
act="relu",
|
||||
name=name + "_branch2b",
|
||||
)
|
||||
self.conv2 = ConvBNLayer(
|
||||
in_channels=out_channels,
|
||||
out_channels=out_channels * 4,
|
||||
kernel_size=1,
|
||||
act=None,
|
||||
name=name + "_branch2c",
|
||||
)
|
||||
|
||||
if not shortcut:
|
||||
self.short = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels * 4,
|
||||
kernel_size=1,
|
||||
stride=stride,
|
||||
is_vd_mode=not if_first and stride[0] != 1,
|
||||
name=name + "_branch1",
|
||||
)
|
||||
|
||||
self.shortcut = shortcut
|
||||
|
||||
def forward(self, inputs):
|
||||
y = self.conv0(inputs)
|
||||
|
||||
conv1 = self.conv1(y)
|
||||
conv2 = self.conv2(conv1)
|
||||
|
||||
if self.shortcut:
|
||||
short = inputs
|
||||
else:
|
||||
short = self.short(inputs)
|
||||
y = paddle.add(x=short, y=conv2)
|
||||
y = F.relu(y)
|
||||
return y
|
||||
|
||||
|
||||
class BasicBlock(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
stride,
|
||||
shortcut=True,
|
||||
if_first=False,
|
||||
name=None,
|
||||
):
|
||||
super(BasicBlock, self).__init__()
|
||||
self.stride = stride
|
||||
self.conv0 = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
act="relu",
|
||||
name=name + "_branch2a",
|
||||
)
|
||||
self.conv1 = ConvBNLayer(
|
||||
in_channels=out_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=3,
|
||||
act=None,
|
||||
name=name + "_branch2b",
|
||||
)
|
||||
|
||||
if not shortcut:
|
||||
self.short = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=1,
|
||||
stride=stride,
|
||||
is_vd_mode=not if_first and stride[0] != 1,
|
||||
name=name + "_branch1",
|
||||
)
|
||||
|
||||
self.shortcut = shortcut
|
||||
|
||||
def forward(self, inputs):
|
||||
y = self.conv0(inputs)
|
||||
conv1 = self.conv1(y)
|
||||
|
||||
if self.shortcut:
|
||||
short = inputs
|
||||
else:
|
||||
short = self.short(inputs)
|
||||
y = paddle.add(x=short, y=conv1)
|
||||
y = F.relu(y)
|
||||
return y
|
||||
|
||||
|
||||
class ResNet(nn.Layer):
|
||||
def __init__(self, in_channels=3, layers=50, **kwargs):
|
||||
super(ResNet, self).__init__()
|
||||
|
||||
self.layers = layers
|
||||
supported_layers = [18, 34, 50, 101, 152, 200]
|
||||
assert (
|
||||
layers in supported_layers
|
||||
), "supported layers are {} but input layer is {}".format(
|
||||
supported_layers, layers
|
||||
)
|
||||
|
||||
if layers == 18:
|
||||
depth = [2, 2, 2, 2]
|
||||
elif layers == 34 or layers == 50:
|
||||
depth = [3, 4, 6, 3]
|
||||
elif layers == 101:
|
||||
depth = [3, 4, 23, 3]
|
||||
elif layers == 152:
|
||||
depth = [3, 8, 36, 3]
|
||||
elif layers == 200:
|
||||
depth = [3, 12, 48, 3]
|
||||
num_channels = [64, 256, 512, 1024] if layers >= 50 else [64, 64, 128, 256]
|
||||
num_filters = [64, 128, 256, 512]
|
||||
|
||||
self.conv1_1 = ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=32,
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
act="relu",
|
||||
name="conv1_1",
|
||||
)
|
||||
self.conv1_2 = ConvBNLayer(
|
||||
in_channels=32,
|
||||
out_channels=32,
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
act="relu",
|
||||
name="conv1_2",
|
||||
)
|
||||
self.conv1_3 = ConvBNLayer(
|
||||
in_channels=32,
|
||||
out_channels=64,
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
act="relu",
|
||||
name="conv1_3",
|
||||
)
|
||||
self.pool2d_max = nn.MaxPool2D(kernel_size=3, stride=2, padding=1)
|
||||
|
||||
self.block_list = []
|
||||
if layers >= 50:
|
||||
for block in range(len(depth)):
|
||||
shortcut = False
|
||||
for i in range(depth[block]):
|
||||
if layers in [101, 152, 200] and block == 2:
|
||||
if i == 0:
|
||||
conv_name = "res" + str(block + 2) + "a"
|
||||
else:
|
||||
conv_name = "res" + str(block + 2) + "b" + str(i)
|
||||
else:
|
||||
conv_name = "res" + str(block + 2) + chr(97 + i)
|
||||
|
||||
if i == 0 and block != 0:
|
||||
stride = (2, 1)
|
||||
else:
|
||||
stride = (1, 1)
|
||||
bottleneck_block = self.add_sublayer(
|
||||
"bb_%d_%d" % (block, i),
|
||||
BottleneckBlock(
|
||||
in_channels=(
|
||||
num_channels[block]
|
||||
if i == 0
|
||||
else num_filters[block] * 4
|
||||
),
|
||||
out_channels=num_filters[block],
|
||||
stride=stride,
|
||||
shortcut=shortcut,
|
||||
if_first=block == i == 0,
|
||||
name=conv_name,
|
||||
),
|
||||
)
|
||||
shortcut = True
|
||||
self.block_list.append(bottleneck_block)
|
||||
self.out_channels = num_filters[block] * 4
|
||||
else:
|
||||
for block in range(len(depth)):
|
||||
shortcut = False
|
||||
for i in range(depth[block]):
|
||||
conv_name = "res" + str(block + 2) + chr(97 + i)
|
||||
if i == 0 and block != 0:
|
||||
stride = (2, 1)
|
||||
else:
|
||||
stride = (1, 1)
|
||||
|
||||
basic_block = self.add_sublayer(
|
||||
"bb_%d_%d" % (block, i),
|
||||
BasicBlock(
|
||||
in_channels=(
|
||||
num_channels[block] if i == 0 else num_filters[block]
|
||||
),
|
||||
out_channels=num_filters[block],
|
||||
stride=stride,
|
||||
shortcut=shortcut,
|
||||
if_first=block == i == 0,
|
||||
name=conv_name,
|
||||
),
|
||||
)
|
||||
shortcut = True
|
||||
self.block_list.append(basic_block)
|
||||
self.out_channels = num_filters[block]
|
||||
self.out_pool = nn.MaxPool2D(kernel_size=2, stride=2, padding=0)
|
||||
|
||||
def forward(self, inputs):
|
||||
y = self.conv1_1(inputs)
|
||||
y = self.conv1_2(y)
|
||||
y = self.conv1_3(y)
|
||||
y = self.pool2d_max(y)
|
||||
for block in self.block_list:
|
||||
y = block(y)
|
||||
y = self.out_pool(y)
|
||||
return y
|
||||
1227
ppocr/modeling/backbones/rec_resnetv2.py
Normal file
1227
ppocr/modeling/backbones/rec_resnetv2.py
Normal file
File diff suppressed because it is too large
Load Diff
82
ppocr/modeling/backbones/rec_shallow_cnn.py
Normal file
82
ppocr/modeling/backbones/rec_shallow_cnn.py
Normal file
@@ -0,0 +1,82 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/open-mmlab/mmocr/blob/1.x/mmocr/models/textrecog/backbones/shallow_cnn.py
|
||||
"""
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import math
|
||||
import numpy as np
|
||||
import paddle
|
||||
from paddle import ParamAttr
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
from paddle.nn import MaxPool2D
|
||||
from paddle.nn.initializer import KaimingNormal, Uniform, Constant
|
||||
|
||||
|
||||
class ConvBNLayer(nn.Layer):
|
||||
def __init__(
|
||||
self, num_channels, filter_size, num_filters, stride, padding, num_groups=1
|
||||
):
|
||||
super(ConvBNLayer, self).__init__()
|
||||
|
||||
self.conv = nn.Conv2D(
|
||||
in_channels=num_channels,
|
||||
out_channels=num_filters,
|
||||
kernel_size=filter_size,
|
||||
stride=stride,
|
||||
padding=padding,
|
||||
groups=num_groups,
|
||||
weight_attr=ParamAttr(initializer=KaimingNormal()),
|
||||
bias_attr=False,
|
||||
)
|
||||
|
||||
self.bn = nn.BatchNorm2D(
|
||||
num_filters,
|
||||
weight_attr=ParamAttr(initializer=Uniform(0, 1)),
|
||||
bias_attr=ParamAttr(initializer=Constant(0)),
|
||||
)
|
||||
self.relu = nn.ReLU()
|
||||
|
||||
def forward(self, inputs):
|
||||
y = self.conv(inputs)
|
||||
y = self.bn(y)
|
||||
y = self.relu(y)
|
||||
return y
|
||||
|
||||
|
||||
class ShallowCNN(nn.Layer):
|
||||
def __init__(self, in_channels=1, hidden_dim=512):
|
||||
super().__init__()
|
||||
assert isinstance(in_channels, int)
|
||||
assert isinstance(hidden_dim, int)
|
||||
|
||||
self.conv1 = ConvBNLayer(in_channels, 3, hidden_dim // 2, stride=1, padding=1)
|
||||
self.conv2 = ConvBNLayer(hidden_dim // 2, 3, hidden_dim, stride=1, padding=1)
|
||||
self.pool = nn.MaxPool2D(kernel_size=2, stride=2, padding=0)
|
||||
self.out_channels = hidden_dim
|
||||
|
||||
def forward(self, x):
|
||||
x = self.conv1(x)
|
||||
x = self.pool(x)
|
||||
|
||||
x = self.conv2(x)
|
||||
x = self.pool(x)
|
||||
|
||||
return x
|
||||
642
ppocr/modeling/backbones/rec_svtrnet.py
Normal file
642
ppocr/modeling/backbones/rec_svtrnet.py
Normal file
@@ -0,0 +1,642 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from paddle import ParamAttr
|
||||
from paddle.nn.initializer import KaimingNormal
|
||||
import numpy as np
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
from paddle.nn.initializer import TruncatedNormal, Constant, Normal
|
||||
|
||||
trunc_normal_ = TruncatedNormal(std=0.02)
|
||||
normal_ = Normal
|
||||
zeros_ = Constant(value=0.0)
|
||||
ones_ = Constant(value=1.0)
|
||||
|
||||
|
||||
def drop_path(x, drop_prob=0.0, training=False):
|
||||
"""Drop paths (Stochastic Depth) per sample (when applied in main path of residual blocks).
|
||||
the original name is misleading as 'Drop Connect' is a different form of dropout in a separate paper...
|
||||
See discussion: https://github.com/tensorflow/tpu/issues/494#issuecomment-532968956 ...
|
||||
"""
|
||||
if drop_prob == 0.0 or not training:
|
||||
return x
|
||||
keep_prob = paddle.to_tensor(1 - drop_prob, dtype=x.dtype)
|
||||
shape = (x.shape[0],) + (1,) * (x.ndim - 1)
|
||||
random_tensor = keep_prob + paddle.rand(shape, dtype=x.dtype)
|
||||
random_tensor = paddle.floor(random_tensor) # binarize
|
||||
output = x.divide(keep_prob) * random_tensor
|
||||
return output
|
||||
|
||||
|
||||
class ConvBNLayer(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=0,
|
||||
bias_attr=False,
|
||||
groups=1,
|
||||
act=nn.GELU,
|
||||
):
|
||||
super().__init__()
|
||||
self.conv = nn.Conv2D(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=kernel_size,
|
||||
stride=stride,
|
||||
padding=padding,
|
||||
groups=groups,
|
||||
weight_attr=paddle.ParamAttr(initializer=nn.initializer.KaimingUniform()),
|
||||
bias_attr=bias_attr,
|
||||
)
|
||||
self.norm = nn.BatchNorm2D(out_channels)
|
||||
self.act = act()
|
||||
|
||||
def forward(self, inputs):
|
||||
out = self.conv(inputs)
|
||||
out = self.norm(out)
|
||||
out = self.act(out)
|
||||
return out
|
||||
|
||||
|
||||
class DropPath(nn.Layer):
|
||||
"""Drop paths (Stochastic Depth) per sample (when applied in main path of residual blocks)."""
|
||||
|
||||
def __init__(self, drop_prob=None):
|
||||
super(DropPath, self).__init__()
|
||||
self.drop_prob = drop_prob
|
||||
|
||||
def forward(self, x):
|
||||
return drop_path(x, self.drop_prob, self.training)
|
||||
|
||||
|
||||
class Identity(nn.Layer):
|
||||
def __init__(self):
|
||||
super(Identity, self).__init__()
|
||||
|
||||
def forward(self, input):
|
||||
return input
|
||||
|
||||
|
||||
class Mlp(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_features,
|
||||
hidden_features=None,
|
||||
out_features=None,
|
||||
act_layer=nn.GELU,
|
||||
drop=0.0,
|
||||
):
|
||||
super().__init__()
|
||||
out_features = out_features or in_features
|
||||
hidden_features = hidden_features or in_features
|
||||
self.fc1 = nn.Linear(in_features, hidden_features)
|
||||
self.act = act_layer()
|
||||
self.fc2 = nn.Linear(hidden_features, out_features)
|
||||
self.drop = nn.Dropout(drop)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.fc1(x)
|
||||
x = self.act(x)
|
||||
x = self.drop(x)
|
||||
x = self.fc2(x)
|
||||
x = self.drop(x)
|
||||
return x
|
||||
|
||||
|
||||
class ConvMixer(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
dim,
|
||||
num_heads=8,
|
||||
HW=[8, 25],
|
||||
local_k=[3, 3],
|
||||
):
|
||||
super().__init__()
|
||||
self.HW = HW
|
||||
self.dim = dim
|
||||
self.local_mixer = nn.Conv2D(
|
||||
dim,
|
||||
dim,
|
||||
local_k,
|
||||
1,
|
||||
[local_k[0] // 2, local_k[1] // 2],
|
||||
groups=num_heads,
|
||||
weight_attr=ParamAttr(initializer=KaimingNormal()),
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
h = self.HW[0]
|
||||
w = self.HW[1]
|
||||
x = x.transpose([0, 2, 1]).reshape([0, self.dim, h, w])
|
||||
x = self.local_mixer(x)
|
||||
x = x.flatten(2).transpose([0, 2, 1])
|
||||
return x
|
||||
|
||||
|
||||
class Attention(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
dim,
|
||||
num_heads=8,
|
||||
mixer="Global",
|
||||
HW=None,
|
||||
local_k=[7, 11],
|
||||
qkv_bias=False,
|
||||
qk_scale=None,
|
||||
attn_drop=0.0,
|
||||
proj_drop=0.0,
|
||||
):
|
||||
super().__init__()
|
||||
self.num_heads = num_heads
|
||||
self.dim = dim
|
||||
self.head_dim = dim // num_heads
|
||||
self.scale = qk_scale or self.head_dim**-0.5
|
||||
|
||||
self.qkv = nn.Linear(dim, dim * 3, bias_attr=qkv_bias)
|
||||
self.attn_drop = nn.Dropout(attn_drop)
|
||||
self.proj = nn.Linear(dim, dim)
|
||||
self.proj_drop = nn.Dropout(proj_drop)
|
||||
self.HW = HW
|
||||
if HW is not None:
|
||||
H = HW[0]
|
||||
W = HW[1]
|
||||
self.N = H * W
|
||||
self.C = dim
|
||||
if mixer == "Local" and HW is not None:
|
||||
hk = local_k[0]
|
||||
wk = local_k[1]
|
||||
mask = paddle.ones([H * W, H + hk - 1, W + wk - 1], dtype="float32")
|
||||
for h in range(0, H):
|
||||
for w in range(0, W):
|
||||
mask[h * W + w, h : h + hk, w : w + wk] = 0.0
|
||||
mask_paddle = mask[:, hk // 2 : H + hk // 2, wk // 2 : W + wk // 2].flatten(
|
||||
1
|
||||
)
|
||||
mask_inf = paddle.full([H * W, H * W], "-inf", dtype="float32")
|
||||
mask = paddle.where(mask_paddle < 1, mask_paddle, mask_inf)
|
||||
self.mask = mask.unsqueeze([0, 1])
|
||||
self.mixer = mixer
|
||||
|
||||
def forward(self, x):
|
||||
qkv = (
|
||||
self.qkv(x)
|
||||
.reshape((0, -1, 3, self.num_heads, self.head_dim))
|
||||
.transpose((2, 0, 3, 1, 4))
|
||||
)
|
||||
q, k, v = qkv[0] * self.scale, qkv[1], qkv[2]
|
||||
|
||||
attn = q.matmul(k.transpose((0, 1, 3, 2)))
|
||||
if self.mixer == "Local":
|
||||
attn += self.mask
|
||||
attn = nn.functional.softmax(attn, axis=-1)
|
||||
attn = self.attn_drop(attn)
|
||||
|
||||
x = (attn.matmul(v)).transpose((0, 2, 1, 3)).reshape((0, -1, self.dim))
|
||||
x = self.proj(x)
|
||||
x = self.proj_drop(x)
|
||||
return x
|
||||
|
||||
|
||||
class Block(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
dim,
|
||||
num_heads,
|
||||
mixer="Global",
|
||||
local_mixer=[7, 11],
|
||||
HW=None,
|
||||
mlp_ratio=4.0,
|
||||
qkv_bias=False,
|
||||
qk_scale=None,
|
||||
drop=0.0,
|
||||
attn_drop=0.0,
|
||||
drop_path=0.0,
|
||||
act_layer=nn.GELU,
|
||||
norm_layer="nn.LayerNorm",
|
||||
epsilon=1e-6,
|
||||
prenorm=True,
|
||||
):
|
||||
super().__init__()
|
||||
if isinstance(norm_layer, str):
|
||||
self.norm1 = eval(norm_layer)(dim, epsilon=epsilon)
|
||||
else:
|
||||
self.norm1 = norm_layer(dim)
|
||||
if mixer == "Global" or mixer == "Local":
|
||||
self.mixer = Attention(
|
||||
dim,
|
||||
num_heads=num_heads,
|
||||
mixer=mixer,
|
||||
HW=HW,
|
||||
local_k=local_mixer,
|
||||
qkv_bias=qkv_bias,
|
||||
qk_scale=qk_scale,
|
||||
attn_drop=attn_drop,
|
||||
proj_drop=drop,
|
||||
)
|
||||
elif mixer == "Conv":
|
||||
self.mixer = ConvMixer(dim, num_heads=num_heads, HW=HW, local_k=local_mixer)
|
||||
else:
|
||||
raise TypeError("The mixer must be one of [Global, Local, Conv]")
|
||||
|
||||
self.drop_path = DropPath(drop_path) if drop_path > 0.0 else Identity()
|
||||
if isinstance(norm_layer, str):
|
||||
self.norm2 = eval(norm_layer)(dim, epsilon=epsilon)
|
||||
else:
|
||||
self.norm2 = norm_layer(dim)
|
||||
mlp_hidden_dim = int(dim * mlp_ratio)
|
||||
self.mlp_ratio = mlp_ratio
|
||||
self.mlp = Mlp(
|
||||
in_features=dim,
|
||||
hidden_features=mlp_hidden_dim,
|
||||
act_layer=act_layer,
|
||||
drop=drop,
|
||||
)
|
||||
self.prenorm = prenorm
|
||||
|
||||
def forward(self, x):
|
||||
if self.prenorm:
|
||||
x = self.norm1(x + self.drop_path(self.mixer(x)))
|
||||
x = self.norm2(x + self.drop_path(self.mlp(x)))
|
||||
else:
|
||||
x = x + self.drop_path(self.mixer(self.norm1(x)))
|
||||
x = x + self.drop_path(self.mlp(self.norm2(x)))
|
||||
return x
|
||||
|
||||
|
||||
class PatchEmbed(nn.Layer):
|
||||
"""Image to Patch Embedding"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
img_size=[32, 100],
|
||||
in_channels=3,
|
||||
embed_dim=768,
|
||||
sub_num=2,
|
||||
patch_size=[4, 4],
|
||||
mode="pope",
|
||||
):
|
||||
super().__init__()
|
||||
num_patches = (img_size[1] // (2**sub_num)) * (img_size[0] // (2**sub_num))
|
||||
self.img_size = img_size
|
||||
self.num_patches = num_patches
|
||||
self.embed_dim = embed_dim
|
||||
self.norm = None
|
||||
if mode == "pope":
|
||||
if sub_num == 2:
|
||||
self.proj = nn.Sequential(
|
||||
ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=embed_dim // 2,
|
||||
kernel_size=3,
|
||||
stride=2,
|
||||
padding=1,
|
||||
act=nn.GELU,
|
||||
bias_attr=None,
|
||||
),
|
||||
ConvBNLayer(
|
||||
in_channels=embed_dim // 2,
|
||||
out_channels=embed_dim,
|
||||
kernel_size=3,
|
||||
stride=2,
|
||||
padding=1,
|
||||
act=nn.GELU,
|
||||
bias_attr=None,
|
||||
),
|
||||
)
|
||||
if sub_num == 3:
|
||||
self.proj = nn.Sequential(
|
||||
ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=embed_dim // 4,
|
||||
kernel_size=3,
|
||||
stride=2,
|
||||
padding=1,
|
||||
act=nn.GELU,
|
||||
bias_attr=None,
|
||||
),
|
||||
ConvBNLayer(
|
||||
in_channels=embed_dim // 4,
|
||||
out_channels=embed_dim // 2,
|
||||
kernel_size=3,
|
||||
stride=2,
|
||||
padding=1,
|
||||
act=nn.GELU,
|
||||
bias_attr=None,
|
||||
),
|
||||
ConvBNLayer(
|
||||
in_channels=embed_dim // 2,
|
||||
out_channels=embed_dim,
|
||||
kernel_size=3,
|
||||
stride=2,
|
||||
padding=1,
|
||||
act=nn.GELU,
|
||||
bias_attr=None,
|
||||
),
|
||||
)
|
||||
elif mode == "linear":
|
||||
self.proj = nn.Conv2D(
|
||||
1, embed_dim, kernel_size=patch_size, stride=patch_size
|
||||
)
|
||||
self.num_patches = (
|
||||
img_size[0] // patch_size[0] * img_size[1] // patch_size[1]
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
B, C, H, W = x.shape
|
||||
assert (
|
||||
H == self.img_size[0] and W == self.img_size[1]
|
||||
), f"Input image size ({H}*{W}) doesn't match model ({self.img_size[0]}*{self.img_size[1]})."
|
||||
x = self.proj(x).flatten(2).transpose((0, 2, 1))
|
||||
return x
|
||||
|
||||
|
||||
class SubSample(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
types="Pool",
|
||||
stride=[2, 1],
|
||||
sub_norm="nn.LayerNorm",
|
||||
act=None,
|
||||
):
|
||||
super().__init__()
|
||||
self.types = types
|
||||
if types == "Pool":
|
||||
self.avgpool = nn.AvgPool2D(
|
||||
kernel_size=[3, 5], stride=stride, padding=[1, 2]
|
||||
)
|
||||
self.maxpool = nn.MaxPool2D(
|
||||
kernel_size=[3, 5], stride=stride, padding=[1, 2]
|
||||
)
|
||||
self.proj = nn.Linear(in_channels, out_channels)
|
||||
else:
|
||||
self.conv = nn.Conv2D(
|
||||
in_channels,
|
||||
out_channels,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
padding=1,
|
||||
weight_attr=ParamAttr(initializer=KaimingNormal()),
|
||||
)
|
||||
self.norm = eval(sub_norm)(out_channels)
|
||||
if act is not None:
|
||||
self.act = act()
|
||||
else:
|
||||
self.act = None
|
||||
|
||||
def forward(self, x):
|
||||
if self.types == "Pool":
|
||||
x1 = self.avgpool(x)
|
||||
x2 = self.maxpool(x)
|
||||
x = (x1 + x2) * 0.5
|
||||
out = self.proj(x.flatten(2).transpose((0, 2, 1)))
|
||||
else:
|
||||
x = self.conv(x)
|
||||
out = x.flatten(2).transpose((0, 2, 1))
|
||||
out = self.norm(out)
|
||||
if self.act is not None:
|
||||
out = self.act(out)
|
||||
|
||||
return out
|
||||
|
||||
|
||||
class SVTRNet(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
img_size=[32, 100],
|
||||
in_channels=3,
|
||||
embed_dim=[64, 128, 256],
|
||||
depth=[3, 6, 3],
|
||||
num_heads=[2, 4, 8],
|
||||
mixer=["Local"] * 6 + ["Global"] * 6, # Local atten, Global atten, Conv
|
||||
local_mixer=[[7, 11], [7, 11], [7, 11]],
|
||||
patch_merging="Conv", # Conv, Pool, None
|
||||
mlp_ratio=4,
|
||||
qkv_bias=True,
|
||||
qk_scale=None,
|
||||
drop_rate=0.0,
|
||||
last_drop=0.1,
|
||||
attn_drop_rate=0.0,
|
||||
drop_path_rate=0.1,
|
||||
norm_layer="nn.LayerNorm",
|
||||
sub_norm="nn.LayerNorm",
|
||||
epsilon=1e-6,
|
||||
out_channels=192,
|
||||
out_char_num=25,
|
||||
block_unit="Block",
|
||||
act="nn.GELU",
|
||||
last_stage=True,
|
||||
sub_num=2,
|
||||
prenorm=True,
|
||||
use_lenhead=False,
|
||||
**kwargs,
|
||||
):
|
||||
super().__init__()
|
||||
self.img_size = img_size
|
||||
self.embed_dim = embed_dim
|
||||
self.out_channels = out_channels
|
||||
self.prenorm = prenorm
|
||||
patch_merging = (
|
||||
None
|
||||
if patch_merging != "Conv" and patch_merging != "Pool"
|
||||
else patch_merging
|
||||
)
|
||||
self.patch_embed = PatchEmbed(
|
||||
img_size=img_size,
|
||||
in_channels=in_channels,
|
||||
embed_dim=embed_dim[0],
|
||||
sub_num=sub_num,
|
||||
)
|
||||
num_patches = self.patch_embed.num_patches
|
||||
self.HW = [img_size[0] // (2**sub_num), img_size[1] // (2**sub_num)]
|
||||
self.pos_embed = self.create_parameter(
|
||||
shape=[1, num_patches, embed_dim[0]], default_initializer=zeros_
|
||||
)
|
||||
self.add_parameter("pos_embed", self.pos_embed)
|
||||
self.pos_drop = nn.Dropout(p=drop_rate)
|
||||
Block_unit = eval(block_unit)
|
||||
|
||||
dpr = np.linspace(0, drop_path_rate, sum(depth))
|
||||
self.blocks1 = nn.LayerList(
|
||||
[
|
||||
Block_unit(
|
||||
dim=embed_dim[0],
|
||||
num_heads=num_heads[0],
|
||||
mixer=mixer[0 : depth[0]][i],
|
||||
HW=self.HW,
|
||||
local_mixer=local_mixer[0],
|
||||
mlp_ratio=mlp_ratio,
|
||||
qkv_bias=qkv_bias,
|
||||
qk_scale=qk_scale,
|
||||
drop=drop_rate,
|
||||
act_layer=eval(act),
|
||||
attn_drop=attn_drop_rate,
|
||||
drop_path=dpr[0 : depth[0]][i],
|
||||
norm_layer=norm_layer,
|
||||
epsilon=epsilon,
|
||||
prenorm=prenorm,
|
||||
)
|
||||
for i in range(depth[0])
|
||||
]
|
||||
)
|
||||
if patch_merging is not None:
|
||||
self.sub_sample1 = SubSample(
|
||||
embed_dim[0],
|
||||
embed_dim[1],
|
||||
sub_norm=sub_norm,
|
||||
stride=[2, 1],
|
||||
types=patch_merging,
|
||||
)
|
||||
HW = [self.HW[0] // 2, self.HW[1]]
|
||||
else:
|
||||
HW = self.HW
|
||||
self.patch_merging = patch_merging
|
||||
self.blocks2 = nn.LayerList(
|
||||
[
|
||||
Block_unit(
|
||||
dim=embed_dim[1],
|
||||
num_heads=num_heads[1],
|
||||
mixer=mixer[depth[0] : depth[0] + depth[1]][i],
|
||||
HW=HW,
|
||||
local_mixer=local_mixer[1],
|
||||
mlp_ratio=mlp_ratio,
|
||||
qkv_bias=qkv_bias,
|
||||
qk_scale=qk_scale,
|
||||
drop=drop_rate,
|
||||
act_layer=eval(act),
|
||||
attn_drop=attn_drop_rate,
|
||||
drop_path=dpr[depth[0] : depth[0] + depth[1]][i],
|
||||
norm_layer=norm_layer,
|
||||
epsilon=epsilon,
|
||||
prenorm=prenorm,
|
||||
)
|
||||
for i in range(depth[1])
|
||||
]
|
||||
)
|
||||
if patch_merging is not None:
|
||||
self.sub_sample2 = SubSample(
|
||||
embed_dim[1],
|
||||
embed_dim[2],
|
||||
sub_norm=sub_norm,
|
||||
stride=[2, 1],
|
||||
types=patch_merging,
|
||||
)
|
||||
HW = [self.HW[0] // 4, self.HW[1]]
|
||||
else:
|
||||
HW = self.HW
|
||||
self.blocks3 = nn.LayerList(
|
||||
[
|
||||
Block_unit(
|
||||
dim=embed_dim[2],
|
||||
num_heads=num_heads[2],
|
||||
mixer=mixer[depth[0] + depth[1] :][i],
|
||||
HW=HW,
|
||||
local_mixer=local_mixer[2],
|
||||
mlp_ratio=mlp_ratio,
|
||||
qkv_bias=qkv_bias,
|
||||
qk_scale=qk_scale,
|
||||
drop=drop_rate,
|
||||
act_layer=eval(act),
|
||||
attn_drop=attn_drop_rate,
|
||||
drop_path=dpr[depth[0] + depth[1] :][i],
|
||||
norm_layer=norm_layer,
|
||||
epsilon=epsilon,
|
||||
prenorm=prenorm,
|
||||
)
|
||||
for i in range(depth[2])
|
||||
]
|
||||
)
|
||||
self.last_stage = last_stage
|
||||
if last_stage:
|
||||
self.avg_pool = nn.AdaptiveAvgPool2D([1, out_char_num])
|
||||
self.last_conv = nn.Conv2D(
|
||||
in_channels=embed_dim[2],
|
||||
out_channels=self.out_channels,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
padding=0,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.hardswish = nn.Hardswish()
|
||||
self.dropout = nn.Dropout(p=last_drop, mode="downscale_in_infer")
|
||||
if not prenorm:
|
||||
self.norm = eval(norm_layer)(embed_dim[-1], epsilon=epsilon)
|
||||
self.use_lenhead = use_lenhead
|
||||
if use_lenhead:
|
||||
self.len_conv = nn.Linear(embed_dim[2], self.out_channels)
|
||||
self.hardswish_len = nn.Hardswish()
|
||||
self.dropout_len = nn.Dropout(p=last_drop, mode="downscale_in_infer")
|
||||
|
||||
trunc_normal_(self.pos_embed)
|
||||
self.apply(self._init_weights)
|
||||
|
||||
def _init_weights(self, m):
|
||||
if isinstance(m, nn.Linear):
|
||||
trunc_normal_(m.weight)
|
||||
if isinstance(m, nn.Linear) and m.bias is not None:
|
||||
zeros_(m.bias)
|
||||
elif isinstance(m, nn.LayerNorm):
|
||||
zeros_(m.bias)
|
||||
ones_(m.weight)
|
||||
|
||||
def forward_features(self, x):
|
||||
x = self.patch_embed(x)
|
||||
x = x + self.pos_embed
|
||||
x = self.pos_drop(x)
|
||||
for blk in self.blocks1:
|
||||
x = blk(x)
|
||||
if self.patch_merging is not None:
|
||||
x = self.sub_sample1(
|
||||
x.transpose([0, 2, 1]).reshape(
|
||||
[0, self.embed_dim[0], self.HW[0], self.HW[1]]
|
||||
)
|
||||
)
|
||||
for blk in self.blocks2:
|
||||
x = blk(x)
|
||||
if self.patch_merging is not None:
|
||||
x = self.sub_sample2(
|
||||
x.transpose([0, 2, 1]).reshape(
|
||||
[0, self.embed_dim[1], self.HW[0] // 2, self.HW[1]]
|
||||
)
|
||||
)
|
||||
for blk in self.blocks3:
|
||||
x = blk(x)
|
||||
if not self.prenorm:
|
||||
x = self.norm(x)
|
||||
return x
|
||||
|
||||
def forward(self, x):
|
||||
x = self.forward_features(x)
|
||||
if self.use_lenhead:
|
||||
len_x = self.len_conv(x.mean(1))
|
||||
len_x = self.dropout_len(self.hardswish_len(len_x))
|
||||
if self.last_stage:
|
||||
if self.patch_merging is not None:
|
||||
h = self.HW[0] // 4
|
||||
else:
|
||||
h = self.HW[0]
|
||||
x = self.avg_pool(
|
||||
x.transpose([0, 2, 1]).reshape([0, self.embed_dim[2], h, self.HW[1]])
|
||||
)
|
||||
x = self.last_conv(x)
|
||||
x = self.hardswish(x)
|
||||
x = self.dropout(x)
|
||||
if self.use_lenhead:
|
||||
return x, len_x
|
||||
return x
|
||||
575
ppocr/modeling/backbones/rec_svtrv2.py
Normal file
575
ppocr/modeling/backbones/rec_svtrv2.py
Normal file
@@ -0,0 +1,575 @@
|
||||
# copyright (c) 2024 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from paddle import ParamAttr
|
||||
from paddle.nn.initializer import KaimingNormal
|
||||
import numpy as np
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
from paddle.nn.initializer import TruncatedNormal, Constant, Normal
|
||||
|
||||
trunc_normal_ = TruncatedNormal(std=0.02)
|
||||
normal_ = Normal
|
||||
zeros_ = Constant(value=0.0)
|
||||
ones_ = Constant(value=1.0)
|
||||
|
||||
|
||||
def drop_path(x, drop_prob=0.0, training=False):
|
||||
"""Drop paths (Stochastic Depth) per sample (when applied in main path of residual blocks).
|
||||
the original name is misleading as 'Drop Connect' is a different form of dropout in a separate paper...
|
||||
See discussion: https://github.com/tensorflow/tpu/issues/494#issuecomment-532968956 ...
|
||||
"""
|
||||
if drop_prob == 0.0 or not training:
|
||||
return x
|
||||
keep_prob = paddle.to_tensor(1 - drop_prob, dtype=x.dtype)
|
||||
shape = (paddle.shape(x)[0],) + (1,) * (x.ndim - 1)
|
||||
random_tensor = keep_prob + paddle.rand(shape, dtype=x.dtype)
|
||||
random_tensor = paddle.floor(random_tensor) # binarize
|
||||
output = x.divide(keep_prob) * random_tensor
|
||||
return output
|
||||
|
||||
|
||||
class DropPath(nn.Layer):
|
||||
"""Drop paths (Stochastic Depth) per sample (when applied in main path of residual blocks)."""
|
||||
|
||||
def __init__(self, drop_prob=None):
|
||||
super(DropPath, self).__init__()
|
||||
self.drop_prob = drop_prob
|
||||
|
||||
def forward(self, x):
|
||||
return drop_path(x, self.drop_prob, self.training)
|
||||
|
||||
|
||||
class Identity(nn.Layer):
|
||||
def __init__(self):
|
||||
super(Identity, self).__init__()
|
||||
|
||||
def forward(self, input):
|
||||
return input
|
||||
|
||||
|
||||
class Mlp(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_features,
|
||||
hidden_features=None,
|
||||
out_features=None,
|
||||
act_layer=nn.GELU,
|
||||
drop=0.0,
|
||||
):
|
||||
super().__init__()
|
||||
out_features = out_features or in_features
|
||||
hidden_features = hidden_features or in_features
|
||||
self.fc1 = nn.Linear(in_features, hidden_features)
|
||||
self.act = act_layer()
|
||||
self.fc2 = nn.Linear(hidden_features, out_features)
|
||||
self.drop = nn.Dropout(drop)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.fc1(x)
|
||||
x = self.act(x)
|
||||
x = self.drop(x)
|
||||
x = self.fc2(x)
|
||||
x = self.drop(x)
|
||||
return x
|
||||
|
||||
|
||||
class ConvBNLayer(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
kernel_size=3,
|
||||
stride=1,
|
||||
padding=0,
|
||||
bias_attr=False,
|
||||
groups=1,
|
||||
act=nn.GELU,
|
||||
):
|
||||
super().__init__()
|
||||
self.conv = nn.Conv2D(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=kernel_size,
|
||||
stride=stride,
|
||||
padding=padding,
|
||||
groups=groups,
|
||||
weight_attr=paddle.ParamAttr(initializer=nn.initializer.KaimingUniform()),
|
||||
bias_attr=bias_attr,
|
||||
)
|
||||
self.norm = nn.BatchNorm2D(out_channels)
|
||||
self.act = act()
|
||||
|
||||
def forward(self, inputs):
|
||||
out = self.conv(inputs)
|
||||
out = self.norm(out)
|
||||
out = self.act(out)
|
||||
return out
|
||||
|
||||
|
||||
class Attention(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
dim,
|
||||
num_heads=8,
|
||||
qkv_bias=False,
|
||||
qk_scale=None,
|
||||
attn_drop=0.0,
|
||||
proj_drop=0.0,
|
||||
):
|
||||
super().__init__()
|
||||
self.num_heads = num_heads
|
||||
self.dim = dim
|
||||
self.head_dim = dim // num_heads
|
||||
self.scale = qk_scale or self.head_dim**-0.5
|
||||
|
||||
self.qkv = nn.Linear(dim, dim * 3, bias_attr=qkv_bias)
|
||||
self.attn_drop = nn.Dropout(attn_drop)
|
||||
self.proj = nn.Linear(dim, dim)
|
||||
self.proj_drop = nn.Dropout(proj_drop)
|
||||
|
||||
def forward(self, x):
|
||||
qkv = (
|
||||
self.qkv(x)
|
||||
.reshape((0, -1, 3, self.num_heads, self.head_dim))
|
||||
.transpose((2, 0, 3, 1, 4))
|
||||
)
|
||||
q, k, v = qkv[0], qkv[1], qkv[2]
|
||||
|
||||
attn = (q.matmul(k.transpose((0, 1, 3, 2)))) * self.scale
|
||||
attn = nn.functional.softmax(attn, axis=-1)
|
||||
attn = self.attn_drop(attn)
|
||||
x = (attn.matmul(v)).transpose((0, 2, 1, 3)).reshape((0, -1, self.dim))
|
||||
x = self.proj(x)
|
||||
x = self.proj_drop(x)
|
||||
return x
|
||||
|
||||
|
||||
class Block(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
dim,
|
||||
num_heads,
|
||||
mlp_ratio=4.0,
|
||||
qkv_bias=False,
|
||||
qk_scale=None,
|
||||
drop=0.0,
|
||||
attn_drop=0.0,
|
||||
drop_path=0.0,
|
||||
act_layer=nn.GELU,
|
||||
norm_layer=nn.LayerNorm,
|
||||
epsilon=1e-6,
|
||||
):
|
||||
super().__init__()
|
||||
self.norm1 = norm_layer(dim, epsilon=epsilon)
|
||||
self.mixer = Attention(
|
||||
dim,
|
||||
num_heads=num_heads,
|
||||
qkv_bias=qkv_bias,
|
||||
qk_scale=qk_scale,
|
||||
attn_drop=attn_drop,
|
||||
proj_drop=drop,
|
||||
)
|
||||
self.drop_path = DropPath(drop_path) if drop_path > 0.0 else Identity()
|
||||
self.norm2 = norm_layer(dim, epsilon=epsilon)
|
||||
mlp_hidden_dim = int(dim * mlp_ratio)
|
||||
self.mlp_ratio = mlp_ratio
|
||||
self.mlp = Mlp(
|
||||
in_features=dim,
|
||||
hidden_features=mlp_hidden_dim,
|
||||
act_layer=act_layer,
|
||||
drop=drop,
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.norm1(x + self.drop_path(self.mixer(x)))
|
||||
x = self.norm2(x + self.drop_path(self.mlp(x)))
|
||||
return x
|
||||
|
||||
|
||||
class ConvBlock(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
dim,
|
||||
num_heads,
|
||||
mlp_ratio=4.0,
|
||||
drop=0.0,
|
||||
drop_path=0.0,
|
||||
act_layer=nn.GELU,
|
||||
norm_layer=nn.LayerNorm,
|
||||
epsilon=1e-6,
|
||||
):
|
||||
super().__init__()
|
||||
mlp_hidden_dim = int(dim * mlp_ratio)
|
||||
self.norm1 = norm_layer(dim, epsilon=epsilon)
|
||||
self.mixer = nn.Conv2D(
|
||||
dim,
|
||||
dim,
|
||||
5,
|
||||
1,
|
||||
2,
|
||||
groups=num_heads,
|
||||
weight_attr=ParamAttr(initializer=KaimingNormal()),
|
||||
)
|
||||
self.drop_path = DropPath(drop_path) if drop_path > 0.0 else Identity()
|
||||
self.norm2 = norm_layer(dim, epsilon=epsilon)
|
||||
self.mlp = Mlp(
|
||||
in_features=dim,
|
||||
hidden_features=mlp_hidden_dim,
|
||||
act_layer=act_layer,
|
||||
drop=drop,
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
C, H, W = x.shape[1:]
|
||||
x = x + self.drop_path(self.mixer(x))
|
||||
x = self.norm1(x.flatten(2).transpose([0, 2, 1]))
|
||||
x = self.norm2(x + self.drop_path(self.mlp(x)))
|
||||
x = x.transpose([0, 2, 1]).reshape([0, C, H, W])
|
||||
return x
|
||||
|
||||
|
||||
class FlattenTranspose(nn.Layer):
|
||||
def forward(self, x):
|
||||
return x.flatten(2).transpose([0, 2, 1])
|
||||
|
||||
|
||||
class SubSample2D(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
stride=[2, 1],
|
||||
):
|
||||
super().__init__()
|
||||
self.conv = nn.Conv2D(
|
||||
in_channels,
|
||||
out_channels,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
padding=1,
|
||||
weight_attr=ParamAttr(initializer=KaimingNormal()),
|
||||
)
|
||||
self.norm = nn.LayerNorm(out_channels)
|
||||
|
||||
def forward(self, x, sz):
|
||||
# print(x.shape)
|
||||
x = self.conv(x)
|
||||
C, H, W = x.shape[1:]
|
||||
x = self.norm(x.flatten(2).transpose([0, 2, 1]))
|
||||
x = x.transpose([0, 2, 1]).reshape([0, C, H, W])
|
||||
return x, [H, W]
|
||||
|
||||
|
||||
class SubSample1D(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels,
|
||||
out_channels,
|
||||
stride=[2, 1],
|
||||
):
|
||||
super().__init__()
|
||||
self.conv = nn.Conv2D(
|
||||
in_channels,
|
||||
out_channels,
|
||||
kernel_size=3,
|
||||
stride=stride,
|
||||
padding=1,
|
||||
weight_attr=ParamAttr(initializer=KaimingNormal()),
|
||||
)
|
||||
self.norm = nn.LayerNorm(out_channels)
|
||||
|
||||
def forward(self, x, sz):
|
||||
C = x.shape[-1]
|
||||
x = x.transpose([0, 2, 1]).reshape([0, C, sz[0], sz[1]])
|
||||
x = self.conv(x)
|
||||
C, H, W = x.shape[1:]
|
||||
x = self.norm(x.flatten(2).transpose([0, 2, 1]))
|
||||
return x, [H, W]
|
||||
|
||||
|
||||
class IdentitySize(nn.Layer):
|
||||
def forward(self, x, sz):
|
||||
return x, sz
|
||||
|
||||
|
||||
class SVTRStage(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
dim=64,
|
||||
out_dim=256,
|
||||
depth=3,
|
||||
mixer=["Local"] * 3,
|
||||
sub_k=[2, 1],
|
||||
num_heads=2,
|
||||
mlp_ratio=4,
|
||||
qkv_bias=True,
|
||||
qk_scale=None,
|
||||
drop_rate=0.0,
|
||||
attn_drop_rate=0.0,
|
||||
drop_path=[0.1] * 3,
|
||||
norm_layer=nn.LayerNorm,
|
||||
act=nn.GELU,
|
||||
eps=1e-6,
|
||||
downsample=None,
|
||||
**kwargs,
|
||||
):
|
||||
super().__init__()
|
||||
self.dim = dim
|
||||
|
||||
conv_block_num = sum([1 if mix == "Conv" else 0 for mix in mixer])
|
||||
blocks = []
|
||||
for i in range(depth):
|
||||
if mixer[i] == "Conv":
|
||||
blocks.append(
|
||||
ConvBlock(
|
||||
dim=dim,
|
||||
num_heads=num_heads,
|
||||
mlp_ratio=mlp_ratio,
|
||||
drop=drop_rate,
|
||||
act_layer=act,
|
||||
drop_path=drop_path[i],
|
||||
norm_layer=norm_layer,
|
||||
epsilon=eps,
|
||||
)
|
||||
)
|
||||
else:
|
||||
blocks.append(
|
||||
Block(
|
||||
dim=dim,
|
||||
num_heads=num_heads,
|
||||
mlp_ratio=mlp_ratio,
|
||||
qkv_bias=qkv_bias,
|
||||
qk_scale=qk_scale,
|
||||
drop=drop_rate,
|
||||
act_layer=act,
|
||||
attn_drop=attn_drop_rate,
|
||||
drop_path=drop_path[i],
|
||||
norm_layer=norm_layer,
|
||||
epsilon=eps,
|
||||
)
|
||||
)
|
||||
if i == conv_block_num - 1 and mixer[-1] != "Conv":
|
||||
blocks.append(FlattenTranspose())
|
||||
self.blocks = nn.Sequential(*blocks)
|
||||
if downsample:
|
||||
if mixer[-1] == "Conv":
|
||||
self.downsample = SubSample2D(dim, out_dim, stride=sub_k)
|
||||
elif mixer[-1] == "Global":
|
||||
self.downsample = SubSample1D(dim, out_dim, stride=sub_k)
|
||||
else:
|
||||
self.downsample = IdentitySize()
|
||||
|
||||
def forward(self, x, sz):
|
||||
x = self.blocks(x)
|
||||
x, sz = self.downsample(x, sz)
|
||||
return x, sz
|
||||
|
||||
|
||||
class ADDPosEmbed(nn.Layer):
|
||||
def __init__(self, feat_max_size=[8, 32], embed_dim=768):
|
||||
super().__init__()
|
||||
pos_embed = paddle.zeros(
|
||||
[1, feat_max_size[0] * feat_max_size[1], embed_dim], dtype=paddle.float32
|
||||
)
|
||||
trunc_normal_(pos_embed)
|
||||
pos_embed = pos_embed.transpose([0, 2, 1]).reshape(
|
||||
[1, embed_dim, feat_max_size[0], feat_max_size[1]]
|
||||
)
|
||||
self.pos_embed = self.create_parameter(
|
||||
[1, embed_dim, feat_max_size[0], feat_max_size[1]]
|
||||
)
|
||||
self.add_parameter("pos_embed", self.pos_embed)
|
||||
self.pos_embed.set_value(pos_embed)
|
||||
|
||||
def forward(self, x):
|
||||
sz = x.shape[2:]
|
||||
x = x + self.pos_embed[:, :, : sz[0], : sz[1]]
|
||||
return x
|
||||
|
||||
|
||||
class POPatchEmbed(nn.Layer):
|
||||
"""Image to Patch Embedding"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
in_channels=3,
|
||||
feat_max_size=[8, 32],
|
||||
embed_dim=768,
|
||||
use_pos_embed=False,
|
||||
flatten=False,
|
||||
):
|
||||
super().__init__()
|
||||
patch_embed = [
|
||||
ConvBNLayer(
|
||||
in_channels=in_channels,
|
||||
out_channels=embed_dim // 2,
|
||||
kernel_size=3,
|
||||
stride=2,
|
||||
padding=1,
|
||||
act=nn.GELU,
|
||||
bias_attr=None,
|
||||
),
|
||||
ConvBNLayer(
|
||||
in_channels=embed_dim // 2,
|
||||
out_channels=embed_dim,
|
||||
kernel_size=3,
|
||||
stride=2,
|
||||
padding=1,
|
||||
act=nn.GELU,
|
||||
bias_attr=None,
|
||||
),
|
||||
]
|
||||
if use_pos_embed:
|
||||
patch_embed.append(ADDPosEmbed(feat_max_size, embed_dim))
|
||||
if flatten:
|
||||
patch_embed.append(FlattenTranspose())
|
||||
self.patch_embed = nn.Sequential(*patch_embed)
|
||||
|
||||
def forward(self, x):
|
||||
sz = x.shape[2:]
|
||||
x = self.patch_embed(x)
|
||||
return x, [sz[0] // 4, sz[1] // 4]
|
||||
|
||||
|
||||
class LastStage(nn.Layer):
|
||||
def __init__(self, in_channels, out_channels, last_drop, out_char_num):
|
||||
super().__init__()
|
||||
self.last_conv = nn.Linear(in_channels, out_channels, bias_attr=False)
|
||||
self.hardswish = nn.Hardswish()
|
||||
self.dropout = nn.Dropout(p=last_drop, mode="downscale_in_infer")
|
||||
|
||||
def forward(self, x, sz):
|
||||
x = x.reshape([0, sz[0], sz[1], x.shape[-1]])
|
||||
x = x.mean(1)
|
||||
x = self.last_conv(x)
|
||||
x = self.hardswish(x)
|
||||
x = self.dropout(x)
|
||||
return x, [1, sz[1]]
|
||||
|
||||
|
||||
class OutPool(nn.Layer):
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
|
||||
def forward(self, x, sz):
|
||||
C = x.shape[-1]
|
||||
x = x.transpose([0, 2, 1]).reshape([0, C, sz[0], sz[1]])
|
||||
x = nn.functional.avg_pool2d(x, [sz[0], 2])
|
||||
return x, [1, sz[1] // 2]
|
||||
|
||||
|
||||
class Feat2D(nn.Layer):
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
|
||||
def forward(self, x, sz):
|
||||
C = x.shape[-1]
|
||||
x = x.transpose([0, 2, 1]).reshape([0, C, sz[0], sz[1]])
|
||||
return x, sz
|
||||
|
||||
|
||||
class SVTRv2(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
max_sz=[32, 128],
|
||||
in_channels=3,
|
||||
out_channels=192,
|
||||
out_char_num=25,
|
||||
depths=[3, 6, 3],
|
||||
dims=[64, 128, 256],
|
||||
mixer=[["Conv"] * 3, ["Conv"] * 3 + ["Global"] * 3, ["Global"] * 3],
|
||||
use_pos_embed=False,
|
||||
sub_k=[[1, 1], [2, 1], [1, 1]],
|
||||
num_heads=[2, 4, 8],
|
||||
mlp_ratio=4,
|
||||
qkv_bias=True,
|
||||
qk_scale=None,
|
||||
drop_rate=0.0,
|
||||
last_drop=0.1,
|
||||
attn_drop_rate=0.0,
|
||||
drop_path_rate=0.1,
|
||||
norm_layer=nn.LayerNorm,
|
||||
act=nn.GELU,
|
||||
last_stage=False,
|
||||
eps=1e-6,
|
||||
use_pool=False,
|
||||
feat2d=False,
|
||||
**kwargs,
|
||||
):
|
||||
super().__init__()
|
||||
num_stages = len(depths)
|
||||
self.num_features = dims[-1]
|
||||
|
||||
feat_max_size = [max_sz[0] // 4, max_sz[1] // 4]
|
||||
self.pope = POPatchEmbed(
|
||||
in_channels=in_channels,
|
||||
feat_max_size=feat_max_size,
|
||||
embed_dim=dims[0],
|
||||
use_pos_embed=use_pos_embed,
|
||||
flatten=mixer[0][0] != "Conv",
|
||||
)
|
||||
|
||||
dpr = np.linspace(0, drop_path_rate, sum(depths)) # stochastic depth decay rule
|
||||
|
||||
self.stages = nn.LayerList()
|
||||
for i_stage in range(num_stages):
|
||||
stage = SVTRStage(
|
||||
dim=dims[i_stage],
|
||||
out_dim=dims[i_stage + 1] if i_stage < num_stages - 1 else 0,
|
||||
depth=depths[i_stage],
|
||||
mixer=mixer[i_stage],
|
||||
sub_k=sub_k[i_stage],
|
||||
num_heads=num_heads[i_stage],
|
||||
mlp_ratio=mlp_ratio,
|
||||
qkv_bias=qkv_bias,
|
||||
qk_scale=qk_scale,
|
||||
drop=drop_rate,
|
||||
attn_drop=attn_drop_rate,
|
||||
drop_path=dpr[sum(depths[:i_stage]) : sum(depths[: i_stage + 1])],
|
||||
norm_layer=norm_layer,
|
||||
act=act,
|
||||
downsample=False if i_stage == num_stages - 1 else True,
|
||||
eps=eps,
|
||||
)
|
||||
self.stages.append(stage)
|
||||
|
||||
self.out_channels = self.num_features
|
||||
self.last_stage = last_stage
|
||||
if last_stage:
|
||||
self.out_channels = out_channels
|
||||
self.stages.append(
|
||||
LastStage(self.num_features, out_channels, last_drop, out_char_num)
|
||||
)
|
||||
if use_pool:
|
||||
self.stages.append(OutPool())
|
||||
|
||||
if feat2d:
|
||||
self.stages.append(Feat2D())
|
||||
self.apply(self._init_weights)
|
||||
|
||||
def _init_weights(self, m):
|
||||
if isinstance(m, nn.Linear):
|
||||
trunc_normal_(m.weight)
|
||||
if isinstance(m, nn.Linear) and m.bias is not None:
|
||||
zeros_(m.bias)
|
||||
elif isinstance(m, nn.LayerNorm):
|
||||
zeros_(m.bias)
|
||||
ones_(m.weight)
|
||||
|
||||
def forward(self, x):
|
||||
x, sz = self.pope(x)
|
||||
for stage in self.stages:
|
||||
x, sz = stage(x, sz)
|
||||
return x
|
||||
616
ppocr/modeling/backbones/rec_vary_vit.py
Normal file
616
ppocr/modeling/backbones/rec_vary_vit.py
Normal file
@@ -0,0 +1,616 @@
|
||||
# copyright (c) 2024 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import math
|
||||
from functools import partial
|
||||
from typing import Optional, Tuple, Type
|
||||
|
||||
import numpy as np
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
from paddle.nn.initializer import (
|
||||
Constant,
|
||||
KaimingUniform,
|
||||
Normal,
|
||||
TruncatedNormal,
|
||||
XavierUniform,
|
||||
)
|
||||
from ppocr.modeling.backbones.rec_donut_swin import DonutSwinModelOutput
|
||||
|
||||
zeros_ = Constant(value=0.0)
|
||||
ones_ = Constant(value=1.0)
|
||||
kaiming_normal_ = KaimingUniform(nonlinearity="relu")
|
||||
trunc_normal_ = TruncatedNormal(std=0.02)
|
||||
xavier_uniform_ = XavierUniform()
|
||||
|
||||
|
||||
class MLPBlock(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
embedding_dim: int,
|
||||
mlp_dim: int,
|
||||
act: Type[nn.Layer] = nn.GELU,
|
||||
) -> None:
|
||||
super().__init__()
|
||||
self.lin1 = nn.Linear(embedding_dim, mlp_dim)
|
||||
self.lin2 = nn.Linear(mlp_dim, embedding_dim)
|
||||
self.act = act()
|
||||
|
||||
def forward(self, x):
|
||||
return self.lin2(self.act(self.lin1(x)))
|
||||
|
||||
|
||||
# From https://github.com/facebookresearch/detectron2/blob/main/detectron2/layers/batch_norm.py # noqa
|
||||
# Itself from https://github.com/facebookresearch/ConvNeXt/blob/d1fa8f6fef0a165b27399986cc2bdacc92777e40/models/convnext.py#L119 # noqa
|
||||
class LayerNorm2d(nn.Layer):
|
||||
def __init__(self, num_channels: int, epsilon: float = 1e-6) -> None:
|
||||
super().__init__()
|
||||
self.weight = paddle.create_parameter([num_channels], dtype="float32")
|
||||
ones_(self.weight)
|
||||
self.bias = paddle.create_parameter([num_channels], dtype="float32")
|
||||
zeros_(self.bias)
|
||||
self.epsilon = epsilon
|
||||
|
||||
def forward(self, x):
|
||||
u = x.mean(1, keepdim=True)
|
||||
s = (x - u).pow(2).mean(1, keepdim=True)
|
||||
x = (x - u) / paddle.sqrt(s + self.epsilon)
|
||||
x = self.weight[:, None, None] * x + self.bias[:, None, None]
|
||||
return x
|
||||
|
||||
|
||||
# This class and its supporting functions below lightly adapted from the ViTDet backbone available at: https://github.com/facebookresearch/detectron2/blob/main/detectron2/modeling/backbone/vit.py # noqa
|
||||
class ImageEncoderViT(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
img_size: int = 1024,
|
||||
patch_size: int = 16,
|
||||
in_chans: int = 3,
|
||||
embed_dim: int = 768,
|
||||
depth: int = 12,
|
||||
num_heads: int = 12,
|
||||
mlp_ratio: float = 4.0,
|
||||
out_chans: int = 256,
|
||||
qkv_bias: bool = True,
|
||||
norm_layer: Type[nn.Layer] = nn.LayerNorm,
|
||||
act_layer: Type[nn.Layer] = nn.GELU,
|
||||
use_abs_pos: bool = True,
|
||||
use_rel_pos: bool = False,
|
||||
rel_pos_zero_init: bool = True,
|
||||
window_size: int = 0,
|
||||
global_attn_indexes: Tuple[int, ...] = (),
|
||||
is_formula: bool = False,
|
||||
) -> None:
|
||||
"""
|
||||
Args:
|
||||
img_size (int): Input image size.
|
||||
patch_size (int): Patch size.
|
||||
in_chans (int): Number of input image channels.
|
||||
embed_dim (int): Patch embedding dimension.
|
||||
depth (int): Depth of ViT.
|
||||
num_heads (int): Number of attention heads in each ViT block.
|
||||
mlp_ratio (float): Ratio of mlp hidden dim to embedding dim.
|
||||
qkv_bias (bool): If True, add a learnable bias to query, key, value.
|
||||
norm_layer (nn.Layer): Normalization layer.
|
||||
act_layer (nn.Layer): Activation layer.
|
||||
use_abs_pos (bool): If True, use absolute positional embeddings.
|
||||
use_rel_pos (bool): If True, add relative positional embeddings to the attention map.
|
||||
rel_pos_zero_init (bool): If True, zero initialize relative positional parameters.
|
||||
window_size (int): Window size for window attention blocks.
|
||||
global_attn_indexes (list): Indexes for blocks using global attention.
|
||||
"""
|
||||
super().__init__()
|
||||
self.img_size = img_size
|
||||
|
||||
self.patch_embed = PatchEmbed(
|
||||
kernel_size=(patch_size, patch_size),
|
||||
stride=(patch_size, patch_size),
|
||||
in_chans=in_chans,
|
||||
embed_dim=embed_dim,
|
||||
)
|
||||
|
||||
self.pos_embed = None
|
||||
if use_abs_pos:
|
||||
# Initialize absolute positional embedding with pretrain image size.
|
||||
self.pos_embed = paddle.create_parameter(
|
||||
shape=(1, img_size // patch_size, img_size // patch_size, embed_dim),
|
||||
dtype="float32",
|
||||
)
|
||||
zeros_(self.pos_embed)
|
||||
|
||||
self.blocks = nn.LayerList()
|
||||
for i in range(depth):
|
||||
block = Block(
|
||||
dim=embed_dim,
|
||||
num_heads=num_heads,
|
||||
mlp_ratio=mlp_ratio,
|
||||
qkv_bias=qkv_bias,
|
||||
norm_layer=norm_layer,
|
||||
act_layer=act_layer,
|
||||
use_rel_pos=use_rel_pos,
|
||||
rel_pos_zero_init=rel_pos_zero_init,
|
||||
window_size=window_size if i not in global_attn_indexes else 0,
|
||||
input_size=(img_size // patch_size, img_size // patch_size),
|
||||
)
|
||||
self.blocks.append(block)
|
||||
|
||||
self.neck = nn.Sequential(
|
||||
nn.Conv2D(
|
||||
embed_dim,
|
||||
out_chans,
|
||||
kernel_size=1,
|
||||
bias_attr=False,
|
||||
),
|
||||
LayerNorm2d(out_chans),
|
||||
nn.Conv2D(
|
||||
out_chans,
|
||||
out_chans,
|
||||
kernel_size=3,
|
||||
padding=1,
|
||||
bias_attr=False,
|
||||
),
|
||||
LayerNorm2d(out_chans),
|
||||
)
|
||||
|
||||
self.net_2 = nn.Conv2D(
|
||||
256, 512, kernel_size=3, stride=2, padding=1, bias_attr=False
|
||||
)
|
||||
self.net_3 = nn.Conv2D(
|
||||
512, 1024, kernel_size=3, stride=2, padding=1, bias_attr=False
|
||||
)
|
||||
self.is_formula = is_formula
|
||||
|
||||
def forward(self, x):
|
||||
x = self.patch_embed(x)
|
||||
if self.pos_embed is not None:
|
||||
x = x + self.pos_embed
|
||||
for blk in self.blocks:
|
||||
x = blk(x)
|
||||
x = self.neck(x.transpose([0, 3, 1, 2]))
|
||||
x = self.net_2(x)
|
||||
if self.is_formula:
|
||||
x = self.net_3(x)
|
||||
return x
|
||||
|
||||
|
||||
class Block(nn.Layer):
|
||||
"""Transformer blocks with support of window attention and residual propagation blocks"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
dim: int,
|
||||
num_heads: int,
|
||||
mlp_ratio: float = 4.0,
|
||||
qkv_bias: bool = True,
|
||||
norm_layer: Type[nn.Layer] = nn.LayerNorm,
|
||||
act_layer: Type[nn.Layer] = nn.GELU,
|
||||
use_rel_pos: bool = False,
|
||||
rel_pos_zero_init: bool = True,
|
||||
window_size: int = 0,
|
||||
input_size: Optional[Tuple[int, int]] = None,
|
||||
) -> None:
|
||||
"""
|
||||
Args:
|
||||
dim (int): Number of input channels.
|
||||
num_heads (int): Number of attention heads in each ViT block.
|
||||
mlp_ratio (float): Ratio of mlp hidden dim to embedding dim.
|
||||
qkv_bias (bool): If True, add a learnable bias to query, key, value.
|
||||
norm_layer (nn.Layer): Normalization layer.
|
||||
act_layer (nn.Layer): Activation layer.
|
||||
use_rel_pos (bool): If True, add relative positional embeddings to the attention map.
|
||||
rel_pos_zero_init (bool): If True, zero initialize relative positional parameters.
|
||||
window_size (int): Window size for window attention blocks. If it equals 0, then
|
||||
use global attention.
|
||||
input_size (tuple(int, int) or None): Input resolution for calculating the relative
|
||||
positional parameter size.
|
||||
"""
|
||||
super().__init__()
|
||||
self.norm1 = norm_layer(dim)
|
||||
self.attn = Attention(
|
||||
dim,
|
||||
num_heads=num_heads,
|
||||
qkv_bias=qkv_bias,
|
||||
use_rel_pos=use_rel_pos,
|
||||
rel_pos_zero_init=rel_pos_zero_init,
|
||||
input_size=input_size if window_size == 0 else (window_size, window_size),
|
||||
)
|
||||
|
||||
self.norm2 = norm_layer(dim)
|
||||
self.mlp = MLPBlock(
|
||||
embedding_dim=dim, mlp_dim=int(dim * mlp_ratio), act=act_layer
|
||||
)
|
||||
|
||||
self.window_size = window_size
|
||||
|
||||
def forward(self, x):
|
||||
shortcut = x
|
||||
|
||||
x = self.norm1(x)
|
||||
# Window partition
|
||||
if self.window_size > 0:
|
||||
H, W = x.shape[1], x.shape[2]
|
||||
x, pad_hw = window_partition(x, self.window_size)
|
||||
x = self.attn(x)
|
||||
# Reverse window partition
|
||||
if self.window_size > 0:
|
||||
x = window_unpartition(x, self.window_size, pad_hw, (H, W))
|
||||
x = shortcut + x
|
||||
x = x + self.mlp(self.norm2(x))
|
||||
|
||||
return x
|
||||
|
||||
|
||||
class Attention(nn.Layer):
|
||||
"""Multi-head Attention block with relative position embeddings."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
dim: int,
|
||||
num_heads: int = 8,
|
||||
qkv_bias: bool = True,
|
||||
use_rel_pos: bool = False,
|
||||
rel_pos_zero_init: bool = True,
|
||||
input_size: Optional[Tuple[int, int]] = None,
|
||||
) -> None:
|
||||
"""
|
||||
Args:
|
||||
dim (int): Number of input channels.
|
||||
num_heads (int): Number of attention heads.
|
||||
qkv_bias (bool): If True, add a learnable bias to query, key, value.
|
||||
rel_pos (bool): If True, add relative positional embeddings to the attention map.
|
||||
rel_pos_zero_init (bool): If True, zero initialize relative positional parameters.
|
||||
input_size (tuple(int, int) or None): Input resolution for calculating the relative
|
||||
positional parameter size.
|
||||
"""
|
||||
super().__init__()
|
||||
self.num_heads = num_heads
|
||||
head_dim = dim // num_heads
|
||||
self.scale = head_dim**-0.5
|
||||
|
||||
self.qkv = nn.Linear(dim, dim * 3, bias_attr=qkv_bias)
|
||||
self.proj = nn.Linear(dim, dim)
|
||||
|
||||
self.use_rel_pos = use_rel_pos
|
||||
if self.use_rel_pos:
|
||||
assert (
|
||||
input_size is not None
|
||||
), "Input size must be provided if using relative positional encoding."
|
||||
# initialize relative positional embeddings
|
||||
self.rel_pos_h = paddle.create_parameter(
|
||||
[2 * input_size[0] - 1, head_dim], dtype="float32"
|
||||
)
|
||||
zeros_(self.rel_pos_h)
|
||||
self.rel_pos_w = paddle.create_parameter(
|
||||
[2 * input_size[1] - 1, head_dim], dtype="float32"
|
||||
)
|
||||
zeros_(self.rel_pos_w)
|
||||
|
||||
def forward(self, x):
|
||||
|
||||
B, H, W, _ = x.shape
|
||||
qkv = (
|
||||
self.qkv(x)
|
||||
.reshape([B, H * W, 3, self.num_heads, -1])
|
||||
.transpose([2, 0, 3, 1, 4])
|
||||
)
|
||||
q, k, v = qkv.reshape([3, B * self.num_heads, H * W, -1]).unbind(0)
|
||||
attn = (q * self.scale) @ k.transpose([0, 2, 1])
|
||||
|
||||
if self.use_rel_pos:
|
||||
attn = add_decomposed_rel_pos(
|
||||
attn, q, self.rel_pos_h, self.rel_pos_w, (H, W), (H, W)
|
||||
)
|
||||
attn = F.softmax(attn, axis=-1)
|
||||
x = (
|
||||
(attn @ v)
|
||||
.reshape([B, self.num_heads, H, W, -1])
|
||||
.transpose([0, 2, 3, 1, 4])
|
||||
.reshape([B, H, W, -1])
|
||||
)
|
||||
x = self.proj(x)
|
||||
|
||||
return x
|
||||
|
||||
|
||||
def window_partition(x, window_size: int):
|
||||
"""
|
||||
Partition into non-overlapping windows with padding if needed.
|
||||
Args:
|
||||
x (tensor): input tokens with [B, H, W, C].
|
||||
window_size (int): window size.
|
||||
|
||||
Returns:
|
||||
windows: windows after partition with [B * num_windows, window_size, window_size, C].
|
||||
(Hp, Wp): padded height and width before partition
|
||||
"""
|
||||
B, H, W, C = x.shape
|
||||
|
||||
pad_h = (window_size - H % window_size) % window_size
|
||||
pad_w = (window_size - W % window_size) % window_size
|
||||
if pad_h > 0 or pad_w > 0:
|
||||
x = F.pad(x, (0, 0, 0, pad_w, 0, pad_h, 0, 0))
|
||||
Hp, Wp = H + pad_h, W + pad_w
|
||||
|
||||
x = x.reshape(
|
||||
[B, Hp // window_size, window_size, Wp // window_size, window_size, C]
|
||||
)
|
||||
windows = x.transpose([0, 1, 3, 2, 4, 5]).reshape([-1, window_size, window_size, C])
|
||||
return windows, (Hp, Wp)
|
||||
|
||||
|
||||
def window_unpartition(
|
||||
windows, window_size: int, pad_hw: Tuple[int, int], hw: Tuple[int, int]
|
||||
):
|
||||
"""
|
||||
Window unpartition into original sequences and removing padding.
|
||||
Args:
|
||||
windows (tensor): input tokens with [B * num_windows, window_size, window_size, C].
|
||||
window_size (int): window size.
|
||||
pad_hw (Tuple): padded height and width (Hp, Wp).
|
||||
hw (Tuple): original height and width (H, W) before padding.
|
||||
|
||||
Returns:
|
||||
x: unpartitioned sequences with [B, H, W, C].
|
||||
"""
|
||||
Hp, Wp = pad_hw
|
||||
H, W = hw
|
||||
B = windows.shape[0] // (Hp * Wp // window_size // window_size)
|
||||
x = windows.reshape(
|
||||
[B, Hp // window_size, Wp // window_size, window_size, window_size, -1]
|
||||
)
|
||||
x = x.transpose([0, 1, 3, 2, 4, 5]).contiguous().reshape([B, Hp, Wp, -1])
|
||||
|
||||
if Hp > H or Wp > W:
|
||||
x = x[:, :H, :W, :].contiguous()
|
||||
return x
|
||||
|
||||
|
||||
def get_rel_pos(q_size: int, k_size: int, rel_pos):
|
||||
"""
|
||||
Get relative positional embeddings according to the relative positions of
|
||||
query and key sizes.
|
||||
Args:
|
||||
q_size (int): size of query q.
|
||||
k_size (int): size of key k.
|
||||
rel_pos (Tensor): relative position embeddings (L, C).
|
||||
|
||||
Returns:
|
||||
Extracted positional embeddings according to relative positions.
|
||||
"""
|
||||
max_rel_dist = int(2 * max(q_size, k_size) - 1)
|
||||
# Interpolate rel pos if needed.
|
||||
if rel_pos.shape[0] != max_rel_dist:
|
||||
# Interpolate rel pos.
|
||||
rel_pos_resized = F.interpolate(
|
||||
rel_pos.reshape(1, rel_pos.shape[0], -1).transpose(0, 2, 1),
|
||||
size=max_rel_dist,
|
||||
mode="linear",
|
||||
)
|
||||
rel_pos_resized = rel_pos_resized.reshape(-1, max_rel_dist).transpose(1, 0)
|
||||
else:
|
||||
rel_pos_resized = rel_pos
|
||||
|
||||
# Scale the coords with short length if shapes for q and k are different.
|
||||
q_coords = paddle.arange(q_size)[:, None] * max(k_size / q_size, 1.0)
|
||||
k_coords = paddle.arange(k_size)[None, :] * max(q_size / k_size, 1.0)
|
||||
relative_coords = (q_coords - k_coords) + (k_size - 1) * max(q_size / k_size, 1.0)
|
||||
|
||||
return rel_pos_resized[relative_coords.cast(paddle.int64)]
|
||||
|
||||
|
||||
def add_decomposed_rel_pos(
|
||||
attn,
|
||||
q,
|
||||
rel_pos_h,
|
||||
rel_pos_w,
|
||||
q_size: Tuple[int, int],
|
||||
k_size: Tuple[int, int],
|
||||
):
|
||||
"""
|
||||
Calculate decomposed Relative Positional Embeddings from :paper:`mvitv2`.
|
||||
https://github.com/facebookresearch/mvit/blob/19786631e330df9f3622e5402b4a419a263a2c80/mvit/models/attention.py # noqa B950
|
||||
Args:
|
||||
attn (Tensor): attention map.
|
||||
q (Tensor): query q in the attention layer with shape (B, q_h * q_w, C).
|
||||
rel_pos_h (Tensor): relative position embeddings (Lh, C) for height axis.
|
||||
rel_pos_w (Tensor): relative position embeddings (Lw, C) for width axis.
|
||||
q_size (Tuple): spatial sequence size of query q with (q_h, q_w).
|
||||
k_size (Tuple): spatial sequence size of key k with (k_h, k_w).
|
||||
|
||||
Returns:
|
||||
attn (Tensor): attention map with added relative positional embeddings.
|
||||
"""
|
||||
q_h, q_w = q_size
|
||||
k_h, k_w = k_size
|
||||
Rh = get_rel_pos(q_h, k_h, rel_pos_h)
|
||||
Rw = get_rel_pos(q_w, k_w, rel_pos_w)
|
||||
|
||||
B, _, dim = q.shape
|
||||
r_q = q.reshape([B, q_h, q_w, dim])
|
||||
rel_h = paddle.einsum("bhwc,hkc->bhwk", r_q, Rh)
|
||||
rel_w = paddle.einsum("bhwc,wkc->bhwk", r_q, Rw)
|
||||
|
||||
attn = (
|
||||
attn.reshape([B, q_h, q_w, k_h, k_w])
|
||||
+ rel_h[:, :, :, :, None]
|
||||
+ rel_w[:, :, :, None, :]
|
||||
).reshape([B, q_h * q_w, k_h * k_w])
|
||||
|
||||
return attn
|
||||
|
||||
|
||||
class PatchEmbed(nn.Layer):
|
||||
"""
|
||||
Image to Patch Embedding.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
kernel_size: Tuple[int, int] = (16, 16),
|
||||
stride: Tuple[int, int] = (16, 16),
|
||||
padding: Tuple[int, int] = (0, 0),
|
||||
in_chans: int = 3,
|
||||
embed_dim: int = 768,
|
||||
) -> None:
|
||||
"""
|
||||
Args:
|
||||
kernel_size (Tuple): kernel size of the projection layer.
|
||||
stride (Tuple): stride of the projection layer.
|
||||
padding (Tuple): padding size of the projection layer.
|
||||
in_chans (int): Number of input image channels.
|
||||
embed_dim (int): Patch embedding dimension.
|
||||
"""
|
||||
super().__init__()
|
||||
|
||||
self.proj = nn.Conv2D(
|
||||
in_chans,
|
||||
embed_dim,
|
||||
kernel_size=kernel_size,
|
||||
stride=stride,
|
||||
padding=padding,
|
||||
weight_attr=True,
|
||||
bias_attr=True,
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.proj(x)
|
||||
# B C H W -> B H W C
|
||||
x = x.transpose([0, 2, 3, 1])
|
||||
return x
|
||||
|
||||
|
||||
def _build_vary(
|
||||
encoder_embed_dim,
|
||||
encoder_depth,
|
||||
encoder_num_heads,
|
||||
encoder_global_attn_indexes,
|
||||
image_size,
|
||||
is_formula=False,
|
||||
):
|
||||
prompt_embed_dim = 256
|
||||
vit_patch_size = 16
|
||||
image_embedding_size = image_size // vit_patch_size
|
||||
image_encoder = ImageEncoderViT(
|
||||
depth=encoder_depth,
|
||||
embed_dim=encoder_embed_dim,
|
||||
img_size=image_size,
|
||||
mlp_ratio=4,
|
||||
norm_layer=partial(paddle.nn.LayerNorm, epsilon=1e-6),
|
||||
num_heads=encoder_num_heads,
|
||||
patch_size=vit_patch_size,
|
||||
qkv_bias=True,
|
||||
use_rel_pos=True,
|
||||
global_attn_indexes=encoder_global_attn_indexes,
|
||||
window_size=14,
|
||||
out_chans=prompt_embed_dim,
|
||||
is_formula=is_formula,
|
||||
)
|
||||
return image_encoder
|
||||
|
||||
|
||||
class Vary_VIT_B(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels=3,
|
||||
image_size=768,
|
||||
encoder_embed_dim=768,
|
||||
encoder_depth=12,
|
||||
encoder_num_heads=12,
|
||||
encoder_global_attn_indexes=[2, 5, 8, 11],
|
||||
):
|
||||
super().__init__()
|
||||
|
||||
self.vision_tower_high = _build_vary(
|
||||
encoder_embed_dim=768,
|
||||
encoder_depth=12,
|
||||
encoder_num_heads=12,
|
||||
encoder_global_attn_indexes=[2, 5, 8, 11],
|
||||
image_size=image_size,
|
||||
)
|
||||
|
||||
self.out_channels = 1024
|
||||
|
||||
def forward(self, input_data):
|
||||
pixel_values = input_data
|
||||
num_channels = pixel_values.shape[1]
|
||||
if num_channels == 1:
|
||||
pixel_values = paddle.repeat_interleave(pixel_values, repeats=3, axis=1)
|
||||
cnn_feature = self.vision_tower_high(pixel_values)
|
||||
cnn_feature = cnn_feature.flatten(2).transpose([0, 2, 1])
|
||||
return cnn_feature
|
||||
|
||||
|
||||
class Vary_VIT_B_Formula(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_channels=3,
|
||||
image_size=768,
|
||||
encoder_embed_dim=768,
|
||||
encoder_depth=12,
|
||||
encoder_num_heads=12,
|
||||
encoder_global_attn_indexes=[2, 5, 8, 11],
|
||||
):
|
||||
"""
|
||||
Vary_VIT_B_Formula
|
||||
Args:
|
||||
in_channels (int): Number of input channels. Default is 3 (for RGB images).
|
||||
image_size (int): Size of the input image. Default is 768.
|
||||
encoder_embed_dim (int): Dimension of the encoder's embedding. Default is 768.
|
||||
encoder_depth (int): Number of layers (depth) in the encoder. Default is 12.
|
||||
encoder_num_heads (int): Number of attention heads in the encoder. Default is 12.
|
||||
encoder_global_attn_indexes (list): List of indices specifying which encoder layers use global attention. Default is [2, 5, 8, 11].
|
||||
Returns:
|
||||
model: nn.Layer. Specific `Vary_VIT_B_Formula` model with defined architecture.
|
||||
"""
|
||||
super(Vary_VIT_B_Formula, self).__init__()
|
||||
|
||||
self.vision_tower_high = _build_vary(
|
||||
encoder_embed_dim=encoder_embed_dim,
|
||||
encoder_depth=encoder_depth,
|
||||
encoder_num_heads=encoder_num_heads,
|
||||
encoder_global_attn_indexes=[2, 5, 8, 11],
|
||||
image_size=image_size,
|
||||
is_formula=True,
|
||||
)
|
||||
self.mm_projector_vary = nn.Linear(1024, 1024)
|
||||
self.out_channels = 1024
|
||||
|
||||
def forward(self, input_data):
|
||||
if self.training:
|
||||
pixel_values, label, attention_mask = input_data
|
||||
else:
|
||||
if isinstance(input_data, list):
|
||||
pixel_values = input_data[0]
|
||||
else:
|
||||
pixel_values = input_data
|
||||
num_channels = pixel_values.shape[1]
|
||||
if num_channels == 1:
|
||||
pixel_values = paddle.repeat_interleave(pixel_values, repeats=3, axis=1)
|
||||
|
||||
cnn_feature = self.vision_tower_high(pixel_values)
|
||||
cnn_feature = cnn_feature.flatten(2).transpose([0, 2, 1])
|
||||
|
||||
cnn_feature = self.mm_projector_vary(cnn_feature)
|
||||
donut_swin_output = DonutSwinModelOutput(
|
||||
last_hidden_state=cnn_feature,
|
||||
pooler_output=None,
|
||||
hidden_states=None,
|
||||
attentions=None,
|
||||
reshaped_hidden_states=None,
|
||||
)
|
||||
if self.training:
|
||||
return donut_swin_output, label, attention_mask
|
||||
else:
|
||||
return donut_swin_output
|
||||
273
ppocr/modeling/backbones/rec_vit.py
Normal file
273
ppocr/modeling/backbones/rec_vit.py
Normal file
@@ -0,0 +1,273 @@
|
||||
# copyright (c) 2023 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from paddle import ParamAttr
|
||||
from paddle.nn.initializer import KaimingNormal
|
||||
import numpy as np
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
from paddle.nn.initializer import TruncatedNormal, Constant, Normal
|
||||
|
||||
trunc_normal_ = TruncatedNormal(std=0.02)
|
||||
normal_ = Normal
|
||||
zeros_ = Constant(value=0.0)
|
||||
ones_ = Constant(value=1.0)
|
||||
|
||||
|
||||
def drop_path(x, drop_prob=0.0, training=False):
|
||||
"""Drop paths (Stochastic Depth) per sample (when applied in main path of residual blocks).
|
||||
the original name is misleading as 'Drop Connect' is a different form of dropout in a separate paper...
|
||||
See discussion: https://github.com/tensorflow/tpu/issues/494#issuecomment-532968956 ...
|
||||
"""
|
||||
if drop_prob == 0.0 or not training:
|
||||
return x
|
||||
keep_prob = paddle.to_tensor(1 - drop_prob)
|
||||
shape = (x.shape[0],) + (1,) * (x.ndim - 1)
|
||||
random_tensor = keep_prob + paddle.rand(shape, dtype=x.dtype)
|
||||
random_tensor = paddle.floor(random_tensor) # binarize
|
||||
output = x.divide(keep_prob) * random_tensor
|
||||
return output
|
||||
|
||||
|
||||
class DropPath(nn.Layer):
|
||||
"""Drop paths (Stochastic Depth) per sample (when applied in main path of residual blocks)."""
|
||||
|
||||
def __init__(self, drop_prob=None):
|
||||
super(DropPath, self).__init__()
|
||||
self.drop_prob = drop_prob
|
||||
|
||||
def forward(self, x):
|
||||
return drop_path(x, self.drop_prob, self.training)
|
||||
|
||||
|
||||
class Identity(nn.Layer):
|
||||
def __init__(self):
|
||||
super(Identity, self).__init__()
|
||||
|
||||
def forward(self, input):
|
||||
return input
|
||||
|
||||
|
||||
class Mlp(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_features,
|
||||
hidden_features=None,
|
||||
out_features=None,
|
||||
act_layer=nn.GELU,
|
||||
drop=0.0,
|
||||
):
|
||||
super().__init__()
|
||||
out_features = out_features or in_features
|
||||
hidden_features = hidden_features or in_features
|
||||
self.fc1 = nn.Linear(in_features, hidden_features)
|
||||
self.act = act_layer()
|
||||
self.fc2 = nn.Linear(hidden_features, out_features)
|
||||
self.drop = nn.Dropout(drop)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.fc1(x)
|
||||
x = self.act(x)
|
||||
x = self.drop(x)
|
||||
x = self.fc2(x)
|
||||
x = self.drop(x)
|
||||
return x
|
||||
|
||||
|
||||
class Attention(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
dim,
|
||||
num_heads=8,
|
||||
qkv_bias=False,
|
||||
qk_scale=None,
|
||||
attn_drop=0.0,
|
||||
proj_drop=0.0,
|
||||
):
|
||||
super().__init__()
|
||||
self.num_heads = num_heads
|
||||
self.dim = dim
|
||||
head_dim = dim // num_heads
|
||||
self.scale = qk_scale or head_dim**-0.5
|
||||
|
||||
self.qkv = nn.Linear(dim, dim * 3, bias_attr=qkv_bias)
|
||||
self.attn_drop = nn.Dropout(attn_drop)
|
||||
self.proj = nn.Linear(dim, dim)
|
||||
self.proj_drop = nn.Dropout(proj_drop)
|
||||
|
||||
def forward(self, x):
|
||||
qkv = paddle.reshape(
|
||||
self.qkv(x), (0, -1, 3, self.num_heads, self.dim // self.num_heads)
|
||||
).transpose((2, 0, 3, 1, 4))
|
||||
q, k, v = qkv[0] * self.scale, qkv[1], qkv[2]
|
||||
|
||||
attn = q.matmul(k.transpose((0, 1, 3, 2)))
|
||||
attn = nn.functional.softmax(attn, axis=-1)
|
||||
attn = self.attn_drop(attn)
|
||||
|
||||
x = (attn.matmul(v)).transpose((0, 2, 1, 3)).reshape((0, -1, self.dim))
|
||||
x = self.proj(x)
|
||||
x = self.proj_drop(x)
|
||||
return x
|
||||
|
||||
|
||||
class Block(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
dim,
|
||||
num_heads,
|
||||
mlp_ratio=4.0,
|
||||
qkv_bias=False,
|
||||
qk_scale=None,
|
||||
drop=0.0,
|
||||
attn_drop=0.0,
|
||||
drop_path=0.0,
|
||||
act_layer=nn.GELU,
|
||||
norm_layer="nn.LayerNorm",
|
||||
epsilon=1e-6,
|
||||
prenorm=True,
|
||||
):
|
||||
super().__init__()
|
||||
if isinstance(norm_layer, str):
|
||||
self.norm1 = eval(norm_layer)(dim, epsilon=epsilon)
|
||||
else:
|
||||
self.norm1 = norm_layer(dim)
|
||||
self.mixer = Attention(
|
||||
dim,
|
||||
num_heads=num_heads,
|
||||
qkv_bias=qkv_bias,
|
||||
qk_scale=qk_scale,
|
||||
attn_drop=attn_drop,
|
||||
proj_drop=drop,
|
||||
)
|
||||
|
||||
self.drop_path = DropPath(drop_path) if drop_path > 0.0 else Identity()
|
||||
if isinstance(norm_layer, str):
|
||||
self.norm2 = eval(norm_layer)(dim, epsilon=epsilon)
|
||||
else:
|
||||
self.norm2 = norm_layer(dim)
|
||||
mlp_hidden_dim = int(dim * mlp_ratio)
|
||||
self.mlp_ratio = mlp_ratio
|
||||
self.mlp = Mlp(
|
||||
in_features=dim,
|
||||
hidden_features=mlp_hidden_dim,
|
||||
act_layer=act_layer,
|
||||
drop=drop,
|
||||
)
|
||||
self.prenorm = prenorm
|
||||
|
||||
def forward(self, x):
|
||||
if self.prenorm:
|
||||
x = self.norm1(x + self.drop_path(self.mixer(x)))
|
||||
x = self.norm2(x + self.drop_path(self.mlp(x)))
|
||||
else:
|
||||
x = x + self.drop_path(self.mixer(self.norm1(x)))
|
||||
x = x + self.drop_path(self.mlp(self.norm2(x)))
|
||||
return x
|
||||
|
||||
|
||||
class ViT(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
img_size=[32, 128],
|
||||
patch_size=[4, 4],
|
||||
in_channels=3,
|
||||
embed_dim=384,
|
||||
depth=12,
|
||||
num_heads=6,
|
||||
mlp_ratio=4,
|
||||
qkv_bias=False,
|
||||
qk_scale=None,
|
||||
drop_rate=0.0,
|
||||
attn_drop_rate=0.0,
|
||||
drop_path_rate=0.1,
|
||||
norm_layer="nn.LayerNorm",
|
||||
epsilon=1e-6,
|
||||
act="nn.GELU",
|
||||
prenorm=False,
|
||||
**kwargs,
|
||||
):
|
||||
super().__init__()
|
||||
self.embed_dim = embed_dim
|
||||
self.out_channels = embed_dim
|
||||
self.prenorm = prenorm
|
||||
self.patch_embed = nn.Conv2D(
|
||||
in_channels, embed_dim, patch_size, patch_size, padding=(0, 0)
|
||||
)
|
||||
self.pos_embed = self.create_parameter(
|
||||
shape=[1, 257, embed_dim], default_initializer=zeros_
|
||||
)
|
||||
self.add_parameter("pos_embed", self.pos_embed)
|
||||
self.pos_drop = nn.Dropout(p=drop_rate)
|
||||
dpr = np.linspace(0, drop_path_rate, depth)
|
||||
self.blocks1 = nn.LayerList(
|
||||
[
|
||||
Block(
|
||||
dim=embed_dim,
|
||||
num_heads=num_heads,
|
||||
mlp_ratio=mlp_ratio,
|
||||
qkv_bias=qkv_bias,
|
||||
qk_scale=qk_scale,
|
||||
drop=drop_rate,
|
||||
act_layer=eval(act),
|
||||
attn_drop=attn_drop_rate,
|
||||
drop_path=dpr[i],
|
||||
norm_layer=norm_layer,
|
||||
epsilon=epsilon,
|
||||
prenorm=prenorm,
|
||||
)
|
||||
for i in range(depth)
|
||||
]
|
||||
)
|
||||
if not prenorm:
|
||||
self.norm = eval(norm_layer)(embed_dim, epsilon=epsilon)
|
||||
|
||||
self.avg_pool = nn.AdaptiveAvgPool2D([1, 25])
|
||||
self.last_conv = nn.Conv2D(
|
||||
in_channels=embed_dim,
|
||||
out_channels=self.out_channels,
|
||||
kernel_size=1,
|
||||
stride=1,
|
||||
padding=0,
|
||||
bias_attr=False,
|
||||
)
|
||||
self.hardswish = nn.Hardswish()
|
||||
self.dropout = nn.Dropout(p=0.1, mode="downscale_in_infer")
|
||||
|
||||
trunc_normal_(self.pos_embed)
|
||||
self.apply(self._init_weights)
|
||||
|
||||
def _init_weights(self, m):
|
||||
if isinstance(m, nn.Linear):
|
||||
trunc_normal_(m.weight)
|
||||
if isinstance(m, nn.Linear) and m.bias is not None:
|
||||
zeros_(m.bias)
|
||||
elif isinstance(m, nn.LayerNorm):
|
||||
zeros_(m.bias)
|
||||
ones_(m.weight)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.patch_embed(x).flatten(2).transpose((0, 2, 1))
|
||||
x = x + self.pos_embed[:, 1:, :] # [:, :x.shape[1], :]
|
||||
x = self.pos_drop(x)
|
||||
for blk in self.blocks1:
|
||||
x = blk(x)
|
||||
if not self.prenorm:
|
||||
x = self.norm(x)
|
||||
|
||||
x = self.avg_pool(x.transpose([0, 2, 1]).reshape([0, self.embed_dim, -1, 25]))
|
||||
x = self.last_conv(x)
|
||||
x = self.hardswish(x)
|
||||
x = self.dropout(x)
|
||||
return x
|
||||
348
ppocr/modeling/backbones/rec_vit_parseq.py
Normal file
348
ppocr/modeling/backbones/rec_vit_parseq.py
Normal file
@@ -0,0 +1,348 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/PaddlePaddle/PaddleClas/blob/release%2F2.5/ppcls/arch/backbone/model_zoo/vision_transformer.py
|
||||
"""
|
||||
|
||||
from collections.abc import Callable
|
||||
|
||||
import numpy as np
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
from paddle.nn.initializer import TruncatedNormal, Constant, Normal
|
||||
|
||||
|
||||
trunc_normal_ = TruncatedNormal(std=0.02)
|
||||
normal_ = Normal
|
||||
zeros_ = Constant(value=0.0)
|
||||
ones_ = Constant(value=1.0)
|
||||
|
||||
|
||||
def to_2tuple(x):
|
||||
return tuple([x] * 2)
|
||||
|
||||
|
||||
def drop_path(x, drop_prob=0.0, training=False):
|
||||
"""Drop paths (Stochastic Depth) per sample (when applied in main path of residual blocks).
|
||||
the original name is misleading as 'Drop Connect' is a different form of dropout in a separate paper...
|
||||
See discussion: https://github.com/tensorflow/tpu/issues/494#issuecomment-532968956 ...
|
||||
"""
|
||||
if drop_prob == 0.0 or not training:
|
||||
return x
|
||||
keep_prob = paddle.to_tensor(1 - drop_prob, dtype=x.dtype)
|
||||
shape = (x.shape[0],) + (1,) * (x.ndim - 1)
|
||||
random_tensor = keep_prob + paddle.rand(shape).astype(x.dtype)
|
||||
random_tensor = paddle.floor(random_tensor) # binarize
|
||||
output = x.divide(keep_prob) * random_tensor
|
||||
return output
|
||||
|
||||
|
||||
class DropPath(nn.Layer):
|
||||
"""Drop paths (Stochastic Depth) per sample (when applied in main path of residual blocks)."""
|
||||
|
||||
def __init__(self, drop_prob=None):
|
||||
super(DropPath, self).__init__()
|
||||
self.drop_prob = drop_prob
|
||||
|
||||
def forward(self, x):
|
||||
return drop_path(x, self.drop_prob, self.training)
|
||||
|
||||
|
||||
class Identity(nn.Layer):
|
||||
def __init__(self):
|
||||
super(Identity, self).__init__()
|
||||
|
||||
def forward(self, input):
|
||||
return input
|
||||
|
||||
|
||||
class Mlp(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
in_features,
|
||||
hidden_features=None,
|
||||
out_features=None,
|
||||
act_layer=nn.GELU,
|
||||
drop=0.0,
|
||||
):
|
||||
super().__init__()
|
||||
out_features = out_features or in_features
|
||||
hidden_features = hidden_features or in_features
|
||||
self.fc1 = nn.Linear(in_features, hidden_features)
|
||||
self.act = act_layer()
|
||||
self.fc2 = nn.Linear(hidden_features, out_features)
|
||||
self.drop = nn.Dropout(drop)
|
||||
|
||||
def forward(self, x):
|
||||
x = self.fc1(x)
|
||||
x = self.act(x)
|
||||
x = self.drop(x)
|
||||
x = self.fc2(x)
|
||||
x = self.drop(x)
|
||||
return x
|
||||
|
||||
|
||||
class Attention(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
dim,
|
||||
num_heads=8,
|
||||
qkv_bias=False,
|
||||
qk_scale=None,
|
||||
attn_drop=0.0,
|
||||
proj_drop=0.0,
|
||||
):
|
||||
super().__init__()
|
||||
self.num_heads = num_heads
|
||||
head_dim = dim // num_heads
|
||||
self.scale = qk_scale or head_dim**-0.5
|
||||
|
||||
self.qkv = nn.Linear(dim, dim * 3, bias_attr=qkv_bias)
|
||||
self.attn_drop = nn.Dropout(attn_drop)
|
||||
self.proj = nn.Linear(dim, dim)
|
||||
self.proj_drop = nn.Dropout(proj_drop)
|
||||
|
||||
def forward(self, x):
|
||||
# B= x.shape[0]
|
||||
N, C = x.shape[1:]
|
||||
qkv = (
|
||||
self.qkv(x)
|
||||
.reshape((-1, N, 3, self.num_heads, C // self.num_heads))
|
||||
.transpose((2, 0, 3, 1, 4))
|
||||
)
|
||||
q, k, v = qkv[0], qkv[1], qkv[2]
|
||||
|
||||
attn = (q.matmul(k.transpose((0, 1, 3, 2)))) * self.scale
|
||||
attn = nn.functional.softmax(attn, axis=-1)
|
||||
attn = self.attn_drop(attn)
|
||||
|
||||
x = (attn.matmul(v)).transpose((0, 2, 1, 3)).reshape((-1, N, C))
|
||||
x = self.proj(x)
|
||||
x = self.proj_drop(x)
|
||||
return x
|
||||
|
||||
|
||||
class Block(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
dim,
|
||||
num_heads,
|
||||
mlp_ratio=4.0,
|
||||
qkv_bias=False,
|
||||
qk_scale=None,
|
||||
drop=0.0,
|
||||
attn_drop=0.0,
|
||||
drop_path=0.0,
|
||||
act_layer=nn.GELU,
|
||||
norm_layer="nn.LayerNorm",
|
||||
epsilon=1e-5,
|
||||
):
|
||||
super().__init__()
|
||||
if isinstance(norm_layer, str):
|
||||
self.norm1 = eval(norm_layer)(dim, epsilon=epsilon)
|
||||
elif isinstance(norm_layer, Callable):
|
||||
self.norm1 = norm_layer(dim)
|
||||
else:
|
||||
raise TypeError("The norm_layer must be str or paddle.nn.layer.Layer class")
|
||||
self.attn = Attention(
|
||||
dim,
|
||||
num_heads=num_heads,
|
||||
qkv_bias=qkv_bias,
|
||||
qk_scale=qk_scale,
|
||||
attn_drop=attn_drop,
|
||||
proj_drop=drop,
|
||||
)
|
||||
# NOTE: drop path for stochastic depth, we shall see if this is better than dropout here
|
||||
self.drop_path = DropPath(drop_path) if drop_path > 0.0 else Identity()
|
||||
if isinstance(norm_layer, str):
|
||||
self.norm2 = eval(norm_layer)(dim, epsilon=epsilon)
|
||||
elif isinstance(norm_layer, Callable):
|
||||
self.norm2 = norm_layer(dim)
|
||||
else:
|
||||
raise TypeError("The norm_layer must be str or paddle.nn.layer.Layer class")
|
||||
mlp_hidden_dim = int(dim * mlp_ratio)
|
||||
self.mlp = Mlp(
|
||||
in_features=dim,
|
||||
hidden_features=mlp_hidden_dim,
|
||||
act_layer=act_layer,
|
||||
drop=drop,
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
x = x + self.drop_path(self.attn(self.norm1(x)))
|
||||
x = x + self.drop_path(self.mlp(self.norm2(x)))
|
||||
return x
|
||||
|
||||
|
||||
class PatchEmbed(nn.Layer):
|
||||
"""Image to Patch Embedding"""
|
||||
|
||||
def __init__(self, img_size=224, patch_size=16, in_chans=3, embed_dim=768):
|
||||
super().__init__()
|
||||
if isinstance(img_size, int):
|
||||
img_size = to_2tuple(img_size)
|
||||
if isinstance(patch_size, int):
|
||||
patch_size = to_2tuple(patch_size)
|
||||
num_patches = (img_size[1] // patch_size[1]) * (img_size[0] // patch_size[0])
|
||||
self.img_size = img_size
|
||||
self.patch_size = patch_size
|
||||
self.num_patches = num_patches
|
||||
|
||||
self.proj = nn.Conv2D(
|
||||
in_chans, embed_dim, kernel_size=patch_size, stride=patch_size
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
B, C, H, W = x.shape
|
||||
assert (
|
||||
H == self.img_size[0] and W == self.img_size[1]
|
||||
), f"Input image size ({H}*{W}) doesn't match model ({self.img_size[0]}*{self.img_size[1]})."
|
||||
|
||||
x = self.proj(x).flatten(2).transpose((0, 2, 1))
|
||||
return x
|
||||
|
||||
|
||||
class VisionTransformer(nn.Layer):
|
||||
"""Vision Transformer with support for patch input"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
img_size=224,
|
||||
patch_size=16,
|
||||
in_channels=3,
|
||||
class_num=1000,
|
||||
embed_dim=768,
|
||||
depth=12,
|
||||
num_heads=12,
|
||||
mlp_ratio=4,
|
||||
qkv_bias=False,
|
||||
qk_scale=None,
|
||||
drop_rate=0.0,
|
||||
attn_drop_rate=0.0,
|
||||
drop_path_rate=0.0,
|
||||
norm_layer="nn.LayerNorm",
|
||||
epsilon=1e-5,
|
||||
**kwargs,
|
||||
):
|
||||
super().__init__()
|
||||
self.class_num = class_num
|
||||
|
||||
self.num_features = self.embed_dim = embed_dim
|
||||
|
||||
self.patch_embed = PatchEmbed(
|
||||
img_size=img_size,
|
||||
patch_size=patch_size,
|
||||
in_chans=in_channels,
|
||||
embed_dim=embed_dim,
|
||||
)
|
||||
num_patches = self.patch_embed.num_patches
|
||||
|
||||
self.pos_embed = self.create_parameter(
|
||||
shape=(1, num_patches, embed_dim), default_initializer=zeros_
|
||||
)
|
||||
self.add_parameter("pos_embed", self.pos_embed)
|
||||
self.cls_token = self.create_parameter(
|
||||
shape=(1, 1, embed_dim), default_initializer=zeros_
|
||||
)
|
||||
self.add_parameter("cls_token", self.cls_token)
|
||||
self.pos_drop = nn.Dropout(p=drop_rate)
|
||||
|
||||
dpr = np.linspace(0, drop_path_rate, depth)
|
||||
|
||||
self.blocks = nn.LayerList(
|
||||
[
|
||||
Block(
|
||||
dim=embed_dim,
|
||||
num_heads=num_heads,
|
||||
mlp_ratio=mlp_ratio,
|
||||
qkv_bias=qkv_bias,
|
||||
qk_scale=qk_scale,
|
||||
drop=drop_rate,
|
||||
attn_drop=attn_drop_rate,
|
||||
drop_path=dpr[i],
|
||||
norm_layer=norm_layer,
|
||||
epsilon=epsilon,
|
||||
)
|
||||
for i in range(depth)
|
||||
]
|
||||
)
|
||||
|
||||
self.norm = eval(norm_layer)(embed_dim, epsilon=epsilon)
|
||||
|
||||
# Classifier head
|
||||
self.head = nn.Linear(embed_dim, class_num) if class_num > 0 else Identity()
|
||||
|
||||
trunc_normal_(self.pos_embed)
|
||||
self.out_channels = embed_dim
|
||||
self.apply(self._init_weights)
|
||||
|
||||
def _init_weights(self, m):
|
||||
if isinstance(m, nn.Linear):
|
||||
trunc_normal_(m.weight)
|
||||
if isinstance(m, nn.Linear) and m.bias is not None:
|
||||
zeros_(m.bias)
|
||||
elif isinstance(m, nn.LayerNorm):
|
||||
zeros_(m.bias)
|
||||
ones_(m.weight)
|
||||
|
||||
def forward_features(self, x):
|
||||
B = x.shape[0]
|
||||
x = self.patch_embed(x)
|
||||
x = x + self.pos_embed
|
||||
x = self.pos_drop(x)
|
||||
for blk in self.blocks:
|
||||
x = blk(x)
|
||||
x = self.norm(x)
|
||||
return x
|
||||
|
||||
def forward(self, x):
|
||||
x = self.forward_features(x)
|
||||
x = self.head(x)
|
||||
return x
|
||||
|
||||
|
||||
class ViTParseQ(VisionTransformer):
|
||||
def __init__(
|
||||
self,
|
||||
img_size=[224, 224],
|
||||
patch_size=[16, 16],
|
||||
in_channels=3,
|
||||
embed_dim=768,
|
||||
depth=12,
|
||||
num_heads=12,
|
||||
mlp_ratio=4.0,
|
||||
qkv_bias=True,
|
||||
drop_rate=0.0,
|
||||
attn_drop_rate=0.0,
|
||||
drop_path_rate=0.0,
|
||||
):
|
||||
super().__init__(
|
||||
img_size,
|
||||
patch_size,
|
||||
in_channels,
|
||||
embed_dim=embed_dim,
|
||||
depth=depth,
|
||||
num_heads=num_heads,
|
||||
mlp_ratio=mlp_ratio,
|
||||
qkv_bias=qkv_bias,
|
||||
drop_rate=drop_rate,
|
||||
attn_drop_rate=attn_drop_rate,
|
||||
drop_path_rate=drop_path_rate,
|
||||
class_num=0,
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
return self.forward_features(x)
|
||||
133
ppocr/modeling/backbones/rec_vitstr.py
Normal file
133
ppocr/modeling/backbones/rec_vitstr.py
Normal file
@@ -0,0 +1,133 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/roatienza/deep-text-recognition-benchmark/blob/master/modules/vitstr.py
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
from ppocr.modeling.backbones.rec_svtrnet import (
|
||||
Block,
|
||||
PatchEmbed,
|
||||
zeros_,
|
||||
trunc_normal_,
|
||||
ones_,
|
||||
)
|
||||
|
||||
scale_dim_heads = {"tiny": [192, 3], "small": [384, 6], "base": [768, 12]}
|
||||
|
||||
|
||||
class ViTSTR(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
img_size=[224, 224],
|
||||
in_channels=1,
|
||||
scale="tiny",
|
||||
seqlen=27,
|
||||
patch_size=[16, 16],
|
||||
embed_dim=None,
|
||||
depth=12,
|
||||
num_heads=None,
|
||||
mlp_ratio=4,
|
||||
qkv_bias=True,
|
||||
qk_scale=None,
|
||||
drop_path_rate=0.0,
|
||||
drop_rate=0.0,
|
||||
attn_drop_rate=0.0,
|
||||
norm_layer="nn.LayerNorm",
|
||||
act_layer="nn.GELU",
|
||||
epsilon=1e-6,
|
||||
out_channels=None,
|
||||
**kwargs,
|
||||
):
|
||||
super().__init__()
|
||||
self.seqlen = seqlen
|
||||
embed_dim = embed_dim if embed_dim is not None else scale_dim_heads[scale][0]
|
||||
num_heads = num_heads if num_heads is not None else scale_dim_heads[scale][1]
|
||||
out_channels = out_channels if out_channels is not None else embed_dim
|
||||
self.patch_embed = PatchEmbed(
|
||||
img_size=img_size,
|
||||
in_channels=in_channels,
|
||||
embed_dim=embed_dim,
|
||||
patch_size=patch_size,
|
||||
mode="linear",
|
||||
)
|
||||
num_patches = self.patch_embed.num_patches
|
||||
|
||||
self.pos_embed = self.create_parameter(
|
||||
shape=[1, num_patches + 1, embed_dim], default_initializer=zeros_
|
||||
)
|
||||
self.add_parameter("pos_embed", self.pos_embed)
|
||||
self.cls_token = self.create_parameter(
|
||||
shape=[1, 1, embed_dim], default_initializer=zeros_
|
||||
)
|
||||
self.add_parameter("cls_token", self.cls_token)
|
||||
|
||||
self.pos_drop = nn.Dropout(p=drop_rate)
|
||||
|
||||
dpr = np.linspace(0, drop_path_rate, depth)
|
||||
self.blocks = nn.LayerList(
|
||||
[
|
||||
Block(
|
||||
dim=embed_dim,
|
||||
num_heads=num_heads,
|
||||
mlp_ratio=mlp_ratio,
|
||||
qkv_bias=qkv_bias,
|
||||
qk_scale=qk_scale,
|
||||
drop=drop_rate,
|
||||
attn_drop=attn_drop_rate,
|
||||
drop_path=dpr[i],
|
||||
norm_layer=norm_layer,
|
||||
act_layer=eval(act_layer),
|
||||
epsilon=epsilon,
|
||||
prenorm=False,
|
||||
)
|
||||
for i in range(depth)
|
||||
]
|
||||
)
|
||||
self.norm = eval(norm_layer)(embed_dim, epsilon=epsilon)
|
||||
|
||||
self.out_channels = out_channels
|
||||
|
||||
trunc_normal_(self.pos_embed)
|
||||
trunc_normal_(self.cls_token)
|
||||
self.apply(self._init_weights)
|
||||
|
||||
def _init_weights(self, m):
|
||||
if isinstance(m, nn.Linear):
|
||||
trunc_normal_(m.weight)
|
||||
if isinstance(m, nn.Linear) and m.bias is not None:
|
||||
zeros_(m.bias)
|
||||
elif isinstance(m, nn.LayerNorm):
|
||||
zeros_(m.bias)
|
||||
ones_(m.weight)
|
||||
|
||||
def forward_features(self, x):
|
||||
B = x.shape[0]
|
||||
x = self.patch_embed(x)
|
||||
cls_tokens = paddle.tile(self.cls_token, repeat_times=[B, 1, 1])
|
||||
x = paddle.concat((cls_tokens, x), axis=1)
|
||||
x = x + self.pos_embed
|
||||
x = self.pos_drop(x)
|
||||
for blk in self.blocks:
|
||||
x = blk(x)
|
||||
x = self.norm(x)
|
||||
return x
|
||||
|
||||
def forward(self, x):
|
||||
x = self.forward_features(x)
|
||||
x = x[:, : self.seqlen]
|
||||
return x.transpose([0, 2, 1]).unsqueeze(2)
|
||||
365
ppocr/modeling/backbones/table_master_resnet.py
Normal file
365
ppocr/modeling/backbones/table_master_resnet.py
Normal file
@@ -0,0 +1,365 @@
|
||||
# Copyright (c) 2022 PaddlePaddle Authors. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/JiaquanYe/TableMASTER-mmocr/blob/master/mmocr/models/textrecog/backbones/table_resnet_extra.py
|
||||
"""
|
||||
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
|
||||
|
||||
class BasicBlock(nn.Layer):
|
||||
expansion = 1
|
||||
|
||||
def __init__(self, inplanes, planes, stride=1, downsample=None, gcb_config=None):
|
||||
super(BasicBlock, self).__init__()
|
||||
self.conv1 = nn.Conv2D(
|
||||
inplanes, planes, kernel_size=3, stride=stride, padding=1, bias_attr=False
|
||||
)
|
||||
self.bn1 = nn.BatchNorm2D(planes, momentum=0.9)
|
||||
self.relu = nn.ReLU()
|
||||
self.conv2 = nn.Conv2D(
|
||||
planes, planes, kernel_size=3, stride=1, padding=1, bias_attr=False
|
||||
)
|
||||
self.bn2 = nn.BatchNorm2D(planes, momentum=0.9)
|
||||
self.downsample = downsample
|
||||
self.stride = stride
|
||||
self.gcb_config = gcb_config
|
||||
|
||||
if self.gcb_config is not None:
|
||||
gcb_ratio = gcb_config["ratio"]
|
||||
gcb_headers = gcb_config["headers"]
|
||||
att_scale = gcb_config["att_scale"]
|
||||
fusion_type = gcb_config["fusion_type"]
|
||||
self.context_block = MultiAspectGCAttention(
|
||||
inplanes=planes,
|
||||
ratio=gcb_ratio,
|
||||
headers=gcb_headers,
|
||||
att_scale=att_scale,
|
||||
fusion_type=fusion_type,
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
residual = x
|
||||
|
||||
out = self.conv1(x)
|
||||
out = self.bn1(out)
|
||||
out = self.relu(out)
|
||||
|
||||
out = self.conv2(out)
|
||||
out = self.bn2(out)
|
||||
|
||||
if self.gcb_config is not None:
|
||||
out = self.context_block(out)
|
||||
|
||||
if self.downsample is not None:
|
||||
residual = self.downsample(x)
|
||||
|
||||
out += residual
|
||||
out = self.relu(out)
|
||||
|
||||
return out
|
||||
|
||||
|
||||
def get_gcb_config(gcb_config, layer):
|
||||
if gcb_config is None or not gcb_config["layers"][layer]:
|
||||
return None
|
||||
else:
|
||||
return gcb_config
|
||||
|
||||
|
||||
class TableResNetExtra(nn.Layer):
|
||||
def __init__(self, layers, in_channels=3, gcb_config=None):
|
||||
assert len(layers) >= 4
|
||||
|
||||
super(TableResNetExtra, self).__init__()
|
||||
self.inplanes = 128
|
||||
self.conv1 = nn.Conv2D(
|
||||
in_channels, 64, kernel_size=3, stride=1, padding=1, bias_attr=False
|
||||
)
|
||||
self.bn1 = nn.BatchNorm2D(64)
|
||||
self.relu1 = nn.ReLU()
|
||||
|
||||
self.conv2 = nn.Conv2D(
|
||||
64, 128, kernel_size=3, stride=1, padding=1, bias_attr=False
|
||||
)
|
||||
self.bn2 = nn.BatchNorm2D(128)
|
||||
self.relu2 = nn.ReLU()
|
||||
|
||||
self.maxpool1 = nn.MaxPool2D(kernel_size=2, stride=2)
|
||||
|
||||
self.layer1 = self._make_layer(
|
||||
BasicBlock,
|
||||
256,
|
||||
layers[0],
|
||||
stride=1,
|
||||
gcb_config=get_gcb_config(gcb_config, 0),
|
||||
)
|
||||
|
||||
self.conv3 = nn.Conv2D(
|
||||
256, 256, kernel_size=3, stride=1, padding=1, bias_attr=False
|
||||
)
|
||||
self.bn3 = nn.BatchNorm2D(256)
|
||||
self.relu3 = nn.ReLU()
|
||||
|
||||
self.maxpool2 = nn.MaxPool2D(kernel_size=2, stride=2)
|
||||
|
||||
self.layer2 = self._make_layer(
|
||||
BasicBlock,
|
||||
256,
|
||||
layers[1],
|
||||
stride=1,
|
||||
gcb_config=get_gcb_config(gcb_config, 1),
|
||||
)
|
||||
|
||||
self.conv4 = nn.Conv2D(
|
||||
256, 256, kernel_size=3, stride=1, padding=1, bias_attr=False
|
||||
)
|
||||
self.bn4 = nn.BatchNorm2D(256)
|
||||
self.relu4 = nn.ReLU()
|
||||
|
||||
self.maxpool3 = nn.MaxPool2D(kernel_size=2, stride=2)
|
||||
|
||||
self.layer3 = self._make_layer(
|
||||
BasicBlock,
|
||||
512,
|
||||
layers[2],
|
||||
stride=1,
|
||||
gcb_config=get_gcb_config(gcb_config, 2),
|
||||
)
|
||||
|
||||
self.conv5 = nn.Conv2D(
|
||||
512, 512, kernel_size=3, stride=1, padding=1, bias_attr=False
|
||||
)
|
||||
self.bn5 = nn.BatchNorm2D(512)
|
||||
self.relu5 = nn.ReLU()
|
||||
|
||||
self.layer4 = self._make_layer(
|
||||
BasicBlock,
|
||||
512,
|
||||
layers[3],
|
||||
stride=1,
|
||||
gcb_config=get_gcb_config(gcb_config, 3),
|
||||
)
|
||||
|
||||
self.conv6 = nn.Conv2D(
|
||||
512, 512, kernel_size=3, stride=1, padding=1, bias_attr=False
|
||||
)
|
||||
self.bn6 = nn.BatchNorm2D(512)
|
||||
self.relu6 = nn.ReLU()
|
||||
|
||||
self.out_channels = [256, 256, 512]
|
||||
|
||||
def _make_layer(self, block, planes, blocks, stride=1, gcb_config=None):
|
||||
downsample = None
|
||||
if stride != 1 or self.inplanes != planes * block.expansion:
|
||||
downsample = nn.Sequential(
|
||||
nn.Conv2D(
|
||||
self.inplanes,
|
||||
planes * block.expansion,
|
||||
kernel_size=1,
|
||||
stride=stride,
|
||||
bias_attr=False,
|
||||
),
|
||||
nn.BatchNorm2D(planes * block.expansion),
|
||||
)
|
||||
|
||||
layers = []
|
||||
layers.append(
|
||||
block(self.inplanes, planes, stride, downsample, gcb_config=gcb_config)
|
||||
)
|
||||
self.inplanes = planes * block.expansion
|
||||
for _ in range(1, blocks):
|
||||
layers.append(block(self.inplanes, planes))
|
||||
|
||||
return nn.Sequential(*layers)
|
||||
|
||||
def forward(self, x):
|
||||
f = []
|
||||
x = self.conv1(x)
|
||||
|
||||
x = self.bn1(x)
|
||||
x = self.relu1(x)
|
||||
|
||||
x = self.conv2(x)
|
||||
x = self.bn2(x)
|
||||
x = self.relu2(x)
|
||||
|
||||
x = self.maxpool1(x)
|
||||
x = self.layer1(x)
|
||||
|
||||
x = self.conv3(x)
|
||||
x = self.bn3(x)
|
||||
x = self.relu3(x)
|
||||
f.append(x)
|
||||
|
||||
x = self.maxpool2(x)
|
||||
x = self.layer2(x)
|
||||
|
||||
x = self.conv4(x)
|
||||
x = self.bn4(x)
|
||||
x = self.relu4(x)
|
||||
f.append(x)
|
||||
|
||||
x = self.maxpool3(x)
|
||||
|
||||
x = self.layer3(x)
|
||||
x = self.conv5(x)
|
||||
x = self.bn5(x)
|
||||
x = self.relu5(x)
|
||||
|
||||
x = self.layer4(x)
|
||||
x = self.conv6(x)
|
||||
x = self.bn6(x)
|
||||
x = self.relu6(x)
|
||||
f.append(x)
|
||||
return f
|
||||
|
||||
|
||||
class MultiAspectGCAttention(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
inplanes,
|
||||
ratio,
|
||||
headers,
|
||||
pooling_type="att",
|
||||
att_scale=False,
|
||||
fusion_type="channel_add",
|
||||
):
|
||||
super(MultiAspectGCAttention, self).__init__()
|
||||
assert pooling_type in ["avg", "att"]
|
||||
|
||||
assert fusion_type in ["channel_add", "channel_mul", "channel_concat"]
|
||||
assert (
|
||||
inplanes % headers == 0 and inplanes >= 8
|
||||
) # inplanes must be divided by headers evenly
|
||||
|
||||
self.headers = headers
|
||||
self.inplanes = inplanes
|
||||
self.ratio = ratio
|
||||
self.planes = int(inplanes * ratio)
|
||||
self.pooling_type = pooling_type
|
||||
self.fusion_type = fusion_type
|
||||
self.att_scale = False
|
||||
|
||||
self.single_header_inplanes = int(inplanes / headers)
|
||||
|
||||
if pooling_type == "att":
|
||||
self.conv_mask = nn.Conv2D(self.single_header_inplanes, 1, kernel_size=1)
|
||||
self.softmax = nn.Softmax(axis=2)
|
||||
else:
|
||||
self.avg_pool = nn.AdaptiveAvgPool2D(1)
|
||||
|
||||
if fusion_type == "channel_add":
|
||||
self.channel_add_conv = nn.Sequential(
|
||||
nn.Conv2D(self.inplanes, self.planes, kernel_size=1),
|
||||
nn.LayerNorm([self.planes, 1, 1]),
|
||||
nn.ReLU(),
|
||||
nn.Conv2D(self.planes, self.inplanes, kernel_size=1),
|
||||
)
|
||||
elif fusion_type == "channel_concat":
|
||||
self.channel_concat_conv = nn.Sequential(
|
||||
nn.Conv2D(self.inplanes, self.planes, kernel_size=1),
|
||||
nn.LayerNorm([self.planes, 1, 1]),
|
||||
nn.ReLU(),
|
||||
nn.Conv2D(self.planes, self.inplanes, kernel_size=1),
|
||||
)
|
||||
# for concat
|
||||
self.cat_conv = nn.Conv2D(2 * self.inplanes, self.inplanes, kernel_size=1)
|
||||
elif fusion_type == "channel_mul":
|
||||
self.channel_mul_conv = nn.Sequential(
|
||||
nn.Conv2D(self.inplanes, self.planes, kernel_size=1),
|
||||
nn.LayerNorm([self.planes, 1, 1]),
|
||||
nn.ReLU(),
|
||||
nn.Conv2D(self.planes, self.inplanes, kernel_size=1),
|
||||
)
|
||||
|
||||
def spatial_pool(self, x):
|
||||
batch, channel, height, width = x.shape
|
||||
if self.pooling_type == "att":
|
||||
# [N*headers, C', H , W] C = headers * C'
|
||||
x = x.reshape(
|
||||
[batch * self.headers, self.single_header_inplanes, height, width]
|
||||
)
|
||||
input_x = x
|
||||
|
||||
# [N*headers, C', H * W] C = headers * C'
|
||||
# input_x = input_x.view(batch, channel, height * width)
|
||||
input_x = input_x.reshape(
|
||||
[batch * self.headers, self.single_header_inplanes, height * width]
|
||||
)
|
||||
|
||||
# [N*headers, 1, C', H * W]
|
||||
input_x = input_x.unsqueeze(1)
|
||||
# [N*headers, 1, H, W]
|
||||
context_mask = self.conv_mask(x)
|
||||
# [N*headers, 1, H * W]
|
||||
context_mask = context_mask.reshape(
|
||||
[batch * self.headers, 1, height * width]
|
||||
)
|
||||
|
||||
# scale variance
|
||||
if self.att_scale and self.headers > 1:
|
||||
context_mask = context_mask / paddle.sqrt(self.single_header_inplanes)
|
||||
|
||||
# [N*headers, 1, H * W]
|
||||
context_mask = self.softmax(context_mask)
|
||||
|
||||
# [N*headers, 1, H * W, 1]
|
||||
context_mask = context_mask.unsqueeze(-1)
|
||||
# [N*headers, 1, C', 1] = [N*headers, 1, C', H * W] * [N*headers, 1, H * W, 1]
|
||||
context = paddle.matmul(input_x, context_mask)
|
||||
|
||||
# [N, headers * C', 1, 1]
|
||||
context = context.reshape(
|
||||
[batch, self.headers * self.single_header_inplanes, 1, 1]
|
||||
)
|
||||
else:
|
||||
# [N, C, 1, 1]
|
||||
context = self.avg_pool(x)
|
||||
|
||||
return context
|
||||
|
||||
def forward(self, x):
|
||||
# [N, C, 1, 1]
|
||||
context = self.spatial_pool(x)
|
||||
|
||||
out = x
|
||||
|
||||
if self.fusion_type == "channel_mul":
|
||||
# [N, C, 1, 1]
|
||||
channel_mul_term = F.sigmoid(self.channel_mul_conv(context))
|
||||
out = out * channel_mul_term
|
||||
elif self.fusion_type == "channel_add":
|
||||
# [N, C, 1, 1]
|
||||
channel_add_term = self.channel_add_conv(context)
|
||||
out = out + channel_add_term
|
||||
else:
|
||||
# [N, C, 1, 1]
|
||||
channel_concat_term = self.channel_concat_conv(context)
|
||||
|
||||
# use concat
|
||||
_, C1, _, _ = channel_concat_term.shape
|
||||
N, C2, H, W = out.shape
|
||||
|
||||
out = paddle.concat(
|
||||
[out, channel_concat_term.expand([-1, -1, H, W])], axis=1
|
||||
)
|
||||
out = self.cat_conv(out)
|
||||
out = F.layer_norm(out, [self.inplanes, H, W])
|
||||
out = F.relu(out)
|
||||
|
||||
return out
|
||||
260
ppocr/modeling/backbones/vqa_layoutlm.py
Normal file
260
ppocr/modeling/backbones/vqa_layoutlm.py
Normal file
@@ -0,0 +1,260 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import os
|
||||
from paddle import nn
|
||||
|
||||
from paddlenlp.transformers import (
|
||||
LayoutXLMModel,
|
||||
LayoutXLMForTokenClassification,
|
||||
LayoutXLMForRelationExtraction,
|
||||
)
|
||||
from paddlenlp.transformers import LayoutLMModel, LayoutLMForTokenClassification
|
||||
from paddlenlp.transformers import (
|
||||
LayoutLMv2Model,
|
||||
LayoutLMv2ForTokenClassification,
|
||||
LayoutLMv2ForRelationExtraction,
|
||||
)
|
||||
from paddlenlp.transformers import AutoModel
|
||||
|
||||
__all__ = ["LayoutXLMForSer", "LayoutLMForSer"]
|
||||
|
||||
pretrained_model_dict = {
|
||||
LayoutXLMModel: {
|
||||
"base": "layoutxlm-base-uncased",
|
||||
"vi": "vi-layoutxlm-base-uncased",
|
||||
},
|
||||
LayoutLMModel: {
|
||||
"base": "layoutlm-base-uncased",
|
||||
},
|
||||
LayoutLMv2Model: {
|
||||
"base": "layoutlmv2-base-uncased",
|
||||
"vi": "vi-layoutlmv2-base-uncased",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
class NLPBaseModel(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
base_model_class,
|
||||
model_class,
|
||||
mode="base",
|
||||
type="ser",
|
||||
pretrained=True,
|
||||
checkpoints=None,
|
||||
**kwargs,
|
||||
):
|
||||
super(NLPBaseModel, self).__init__()
|
||||
if checkpoints is not None: # load the trained model
|
||||
self.model = model_class.from_pretrained(checkpoints)
|
||||
else: # load the pretrained-model
|
||||
pretrained_model_name = pretrained_model_dict[base_model_class][mode]
|
||||
if type == "ser":
|
||||
self.model = model_class.from_pretrained(
|
||||
pretrained_model_name, num_classes=kwargs["num_classes"], dropout=0
|
||||
)
|
||||
else:
|
||||
self.model = model_class.from_pretrained(
|
||||
pretrained_model_name, dropout=0
|
||||
)
|
||||
self.out_channels = 1
|
||||
self.use_visual_backbone = True
|
||||
|
||||
|
||||
class LayoutLMForSer(NLPBaseModel):
|
||||
def __init__(
|
||||
self, num_classes, pretrained=True, checkpoints=None, mode="base", **kwargs
|
||||
):
|
||||
super(LayoutLMForSer, self).__init__(
|
||||
LayoutLMModel,
|
||||
LayoutLMForTokenClassification,
|
||||
mode,
|
||||
"ser",
|
||||
pretrained,
|
||||
checkpoints,
|
||||
num_classes=num_classes,
|
||||
)
|
||||
self.use_visual_backbone = False
|
||||
|
||||
def forward(self, x):
|
||||
x = self.model(
|
||||
input_ids=x[0],
|
||||
bbox=x[1],
|
||||
attention_mask=x[2],
|
||||
token_type_ids=x[3],
|
||||
position_ids=None,
|
||||
output_hidden_states=False,
|
||||
)
|
||||
return x
|
||||
|
||||
|
||||
class LayoutLMv2ForSer(NLPBaseModel):
|
||||
def __init__(
|
||||
self, num_classes, pretrained=True, checkpoints=None, mode="base", **kwargs
|
||||
):
|
||||
super(LayoutLMv2ForSer, self).__init__(
|
||||
LayoutLMv2Model,
|
||||
LayoutLMv2ForTokenClassification,
|
||||
mode,
|
||||
"ser",
|
||||
pretrained,
|
||||
checkpoints,
|
||||
num_classes=num_classes,
|
||||
)
|
||||
if (
|
||||
hasattr(self.model.layoutlmv2, "use_visual_backbone")
|
||||
and self.model.layoutlmv2.use_visual_backbone is False
|
||||
):
|
||||
self.use_visual_backbone = False
|
||||
|
||||
def forward(self, x):
|
||||
if self.use_visual_backbone is True:
|
||||
image = x[4]
|
||||
else:
|
||||
image = None
|
||||
x = self.model(
|
||||
input_ids=x[0],
|
||||
bbox=x[1],
|
||||
attention_mask=x[2],
|
||||
token_type_ids=x[3],
|
||||
image=image,
|
||||
position_ids=None,
|
||||
head_mask=None,
|
||||
labels=None,
|
||||
)
|
||||
if self.training:
|
||||
res = {"backbone_out": x[0]}
|
||||
res.update(x[1])
|
||||
return res
|
||||
else:
|
||||
return x
|
||||
|
||||
|
||||
class LayoutXLMForSer(NLPBaseModel):
|
||||
def __init__(
|
||||
self, num_classes, pretrained=True, checkpoints=None, mode="base", **kwargs
|
||||
):
|
||||
super(LayoutXLMForSer, self).__init__(
|
||||
LayoutXLMModel,
|
||||
LayoutXLMForTokenClassification,
|
||||
mode,
|
||||
"ser",
|
||||
pretrained,
|
||||
checkpoints,
|
||||
num_classes=num_classes,
|
||||
)
|
||||
if (
|
||||
hasattr(self.model.layoutxlm, "use_visual_backbone")
|
||||
and self.model.layoutxlm.use_visual_backbone is False
|
||||
):
|
||||
self.use_visual_backbone = False
|
||||
|
||||
def forward(self, x):
|
||||
if self.use_visual_backbone is True:
|
||||
image = x[4]
|
||||
else:
|
||||
image = None
|
||||
x = self.model(
|
||||
input_ids=x[0],
|
||||
bbox=x[1],
|
||||
attention_mask=x[2],
|
||||
token_type_ids=x[3],
|
||||
image=image,
|
||||
position_ids=None,
|
||||
head_mask=None,
|
||||
labels=None,
|
||||
)
|
||||
if self.training:
|
||||
res = {"backbone_out": x[0]}
|
||||
res.update(x[1])
|
||||
return res
|
||||
else:
|
||||
return x
|
||||
|
||||
|
||||
class LayoutLMv2ForRe(NLPBaseModel):
|
||||
def __init__(self, pretrained=True, checkpoints=None, mode="base", **kwargs):
|
||||
super(LayoutLMv2ForRe, self).__init__(
|
||||
LayoutLMv2Model,
|
||||
LayoutLMv2ForRelationExtraction,
|
||||
mode,
|
||||
"re",
|
||||
pretrained,
|
||||
checkpoints,
|
||||
)
|
||||
if (
|
||||
hasattr(self.model.layoutlmv2, "use_visual_backbone")
|
||||
and self.model.layoutlmv2.use_visual_backbone is False
|
||||
):
|
||||
self.use_visual_backbone = False
|
||||
|
||||
def forward(self, x):
|
||||
x = self.model(
|
||||
input_ids=x[0],
|
||||
bbox=x[1],
|
||||
attention_mask=x[2],
|
||||
token_type_ids=x[3],
|
||||
image=x[4],
|
||||
position_ids=None,
|
||||
head_mask=None,
|
||||
labels=None,
|
||||
entities=x[5],
|
||||
relations=x[6],
|
||||
)
|
||||
return x
|
||||
|
||||
|
||||
class LayoutXLMForRe(NLPBaseModel):
|
||||
def __init__(self, pretrained=True, checkpoints=None, mode="base", **kwargs):
|
||||
super(LayoutXLMForRe, self).__init__(
|
||||
LayoutXLMModel,
|
||||
LayoutXLMForRelationExtraction,
|
||||
mode,
|
||||
"re",
|
||||
pretrained,
|
||||
checkpoints,
|
||||
)
|
||||
if (
|
||||
hasattr(self.model.layoutxlm, "use_visual_backbone")
|
||||
and self.model.layoutxlm.use_visual_backbone is False
|
||||
):
|
||||
self.use_visual_backbone = False
|
||||
|
||||
def forward(self, x):
|
||||
if self.use_visual_backbone is True:
|
||||
image = x[4]
|
||||
entities = x[5]
|
||||
relations = x[6]
|
||||
else:
|
||||
image = None
|
||||
entities = x[4]
|
||||
relations = x[5]
|
||||
x = self.model(
|
||||
input_ids=x[0],
|
||||
bbox=x[1],
|
||||
attention_mask=x[2],
|
||||
token_type_ids=x[3],
|
||||
image=image,
|
||||
position_ids=None,
|
||||
head_mask=None,
|
||||
labels=None,
|
||||
entities=entities,
|
||||
relations=relations,
|
||||
)
|
||||
return x
|
||||
Reference in New Issue
Block a user