This commit is contained in:
124
ppocr/losses/__init__.py
Normal file
124
ppocr/losses/__init__.py
Normal file
@@ -0,0 +1,124 @@
|
||||
# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import copy
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
|
||||
# basic_loss
|
||||
from .basic_loss import LossFromOutput
|
||||
|
||||
# det loss
|
||||
from .det_db_loss import DBLoss
|
||||
from .det_east_loss import EASTLoss
|
||||
from .det_sast_loss import SASTLoss
|
||||
from .det_pse_loss import PSELoss
|
||||
from .det_fce_loss import FCELoss
|
||||
from .det_ct_loss import CTLoss
|
||||
from .det_drrg_loss import DRRGLoss
|
||||
|
||||
# rec loss
|
||||
from .rec_ctc_loss import CTCLoss
|
||||
from .rec_att_loss import AttentionLoss
|
||||
from .rec_srn_loss import SRNLoss
|
||||
from .rec_ce_loss import CELoss
|
||||
from .rec_sar_loss import SARLoss
|
||||
from .rec_aster_loss import AsterLoss
|
||||
from .rec_pren_loss import PRENLoss
|
||||
from .rec_multi_loss import MultiLoss
|
||||
from .rec_vl_loss import VLLoss
|
||||
from .rec_spin_att_loss import SPINAttentionLoss
|
||||
from .rec_rfl_loss import RFLLoss
|
||||
from .rec_can_loss import CANLoss
|
||||
from .rec_satrn_loss import SATRNLoss
|
||||
from .rec_nrtr_loss import NRTRLoss
|
||||
from .rec_parseq_loss import ParseQLoss
|
||||
from .rec_cppd_loss import CPPDLoss
|
||||
from .rec_latexocr_loss import LaTeXOCRLoss
|
||||
from .rec_unimernet_loss import UniMERNetLoss
|
||||
from .rec_ppformulanet_loss import PPFormulaNet_S_Loss, PPFormulaNet_L_Loss
|
||||
|
||||
# cls loss
|
||||
from .cls_loss import ClsLoss
|
||||
|
||||
# e2e loss
|
||||
from .e2e_pg_loss import PGLoss
|
||||
from .kie_sdmgr_loss import SDMGRLoss
|
||||
|
||||
# basic loss function
|
||||
from .basic_loss import DistanceLoss
|
||||
|
||||
# combined loss function
|
||||
from .combined_loss import CombinedLoss
|
||||
|
||||
# table loss
|
||||
from .table_att_loss import TableAttentionLoss, SLALoss
|
||||
from .table_master_loss import TableMasterLoss
|
||||
|
||||
# vqa token loss
|
||||
from .vqa_token_layoutlm_loss import VQASerTokenLayoutLMLoss
|
||||
|
||||
# sr loss
|
||||
from .stroke_focus_loss import StrokeFocusLoss
|
||||
from .text_focus_loss import TelescopeLoss
|
||||
|
||||
|
||||
def build_loss(config):
|
||||
support_dict = [
|
||||
"DBLoss",
|
||||
"PSELoss",
|
||||
"EASTLoss",
|
||||
"SASTLoss",
|
||||
"FCELoss",
|
||||
"CTCLoss",
|
||||
"ClsLoss",
|
||||
"AttentionLoss",
|
||||
"SRNLoss",
|
||||
"PGLoss",
|
||||
"CombinedLoss",
|
||||
"CELoss",
|
||||
"TableAttentionLoss",
|
||||
"SARLoss",
|
||||
"AsterLoss",
|
||||
"SDMGRLoss",
|
||||
"VQASerTokenLayoutLMLoss",
|
||||
"LossFromOutput",
|
||||
"PRENLoss",
|
||||
"MultiLoss",
|
||||
"TableMasterLoss",
|
||||
"SPINAttentionLoss",
|
||||
"VLLoss",
|
||||
"StrokeFocusLoss",
|
||||
"SLALoss",
|
||||
"CTLoss",
|
||||
"RFLLoss",
|
||||
"DRRGLoss",
|
||||
"CANLoss",
|
||||
"TelescopeLoss",
|
||||
"SATRNLoss",
|
||||
"NRTRLoss",
|
||||
"ParseQLoss",
|
||||
"CPPDLoss",
|
||||
"LaTeXOCRLoss",
|
||||
"UniMERNetLoss",
|
||||
"PPFormulaNet_S_Loss",
|
||||
"PPFormulaNet_L_Loss",
|
||||
]
|
||||
config = copy.deepcopy(config)
|
||||
module_name = config.pop("name")
|
||||
assert module_name in support_dict, Exception(
|
||||
"loss only support {}".format(support_dict)
|
||||
)
|
||||
module_class = eval(module_name)(**config)
|
||||
return module_class
|
||||
49
ppocr/losses/ace_loss.py
Normal file
49
ppocr/losses/ace_loss.py
Normal file
@@ -0,0 +1,49 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
# This code is refer from: https://github.com/viig99/LS-ACELoss
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
|
||||
|
||||
class ACELoss(nn.Layer):
|
||||
def __init__(self, **kwargs):
|
||||
super().__init__()
|
||||
self.loss_func = nn.CrossEntropyLoss(
|
||||
weight=None, ignore_index=0, reduction="none", soft_label=True, axis=-1
|
||||
)
|
||||
|
||||
def __call__(self, predicts, batch):
|
||||
if isinstance(predicts, (list, tuple)):
|
||||
predicts = predicts[-1]
|
||||
|
||||
B, N = predicts.shape[:2]
|
||||
div = paddle.to_tensor([N]).astype("float32")
|
||||
|
||||
predicts = nn.functional.softmax(predicts, axis=-1)
|
||||
aggregation_preds = paddle.sum(predicts, axis=1)
|
||||
aggregation_preds = paddle.divide(aggregation_preds, div)
|
||||
|
||||
length = batch[2].astype("float32")
|
||||
batch = batch[3].astype("float32")
|
||||
batch[:, 0] = paddle.subtract(div, length)
|
||||
batch = paddle.divide(batch, div)
|
||||
|
||||
loss = self.loss_func(aggregation_preds, batch)
|
||||
return {"loss_ace": loss}
|
||||
247
ppocr/losses/basic_loss.py
Normal file
247
ppocr/losses/basic_loss.py
Normal file
@@ -0,0 +1,247 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
|
||||
from paddle.nn import L1Loss
|
||||
from paddle.nn import MSELoss as L2Loss
|
||||
from paddle.nn import SmoothL1Loss
|
||||
|
||||
|
||||
class CELoss(nn.Layer):
|
||||
def __init__(self, epsilon=None):
|
||||
super().__init__()
|
||||
if epsilon is not None and (epsilon <= 0 or epsilon >= 1):
|
||||
epsilon = None
|
||||
self.epsilon = epsilon
|
||||
|
||||
def _labelsmoothing(self, target, class_num):
|
||||
if target.shape[-1] != class_num:
|
||||
one_hot_target = F.one_hot(target, class_num)
|
||||
else:
|
||||
one_hot_target = target
|
||||
soft_target = F.label_smooth(one_hot_target, epsilon=self.epsilon)
|
||||
soft_target = paddle.reshape(soft_target, shape=[-1, class_num])
|
||||
return soft_target
|
||||
|
||||
def forward(self, x, label):
|
||||
loss_dict = {}
|
||||
if self.epsilon is not None:
|
||||
class_num = x.shape[-1]
|
||||
label = self._labelsmoothing(label, class_num)
|
||||
x = -F.log_softmax(x, axis=-1)
|
||||
loss = paddle.sum(x * label, axis=-1)
|
||||
else:
|
||||
if label.shape[-1] == x.shape[-1]:
|
||||
label = F.softmax(label, axis=-1)
|
||||
soft_label = True
|
||||
else:
|
||||
soft_label = False
|
||||
loss = F.cross_entropy(x, label=label, soft_label=soft_label)
|
||||
return loss
|
||||
|
||||
|
||||
class KLJSLoss(object):
|
||||
def __init__(self, mode="kl"):
|
||||
assert mode in [
|
||||
"kl",
|
||||
"js",
|
||||
"KL",
|
||||
"JS",
|
||||
], "mode can only be one of ['kl', 'KL', 'js', 'JS']"
|
||||
self.mode = mode
|
||||
|
||||
def __call__(self, p1, p2, reduction="mean", eps=1e-5):
|
||||
if self.mode.lower() == "kl":
|
||||
loss = paddle.multiply(p2, paddle.log((p2 + eps) / (p1 + eps) + eps))
|
||||
loss += paddle.multiply(p1, paddle.log((p1 + eps) / (p2 + eps) + eps))
|
||||
loss *= 0.5
|
||||
elif self.mode.lower() == "js":
|
||||
loss = paddle.multiply(
|
||||
p2, paddle.log((2 * p2 + eps) / (p1 + p2 + eps) + eps)
|
||||
)
|
||||
loss += paddle.multiply(
|
||||
p1, paddle.log((2 * p1 + eps) / (p1 + p2 + eps) + eps)
|
||||
)
|
||||
loss *= 0.5
|
||||
else:
|
||||
raise ValueError(
|
||||
"The mode.lower() if KLJSLoss should be one of ['kl', 'js']"
|
||||
)
|
||||
|
||||
if reduction == "mean":
|
||||
loss = paddle.mean(loss, axis=[1, 2])
|
||||
elif reduction == "none" or reduction is None:
|
||||
return loss
|
||||
else:
|
||||
loss = paddle.sum(loss, axis=[1, 2])
|
||||
|
||||
return loss
|
||||
|
||||
|
||||
class DMLLoss(nn.Layer):
|
||||
"""
|
||||
DMLLoss
|
||||
"""
|
||||
|
||||
def __init__(self, act=None, use_log=False):
|
||||
super().__init__()
|
||||
if act is not None:
|
||||
assert act in ["softmax", "sigmoid"]
|
||||
if act == "softmax":
|
||||
self.act = nn.Softmax(axis=-1)
|
||||
elif act == "sigmoid":
|
||||
self.act = nn.Sigmoid()
|
||||
else:
|
||||
self.act = None
|
||||
|
||||
self.use_log = use_log
|
||||
self.jskl_loss = KLJSLoss(mode="kl")
|
||||
|
||||
def _kldiv(self, x, target):
|
||||
eps = 1.0e-10
|
||||
loss = target * (paddle.log(target + eps) - x)
|
||||
# batch mean loss
|
||||
loss = paddle.sum(loss) / loss.shape[0]
|
||||
return loss
|
||||
|
||||
def forward(self, out1, out2):
|
||||
if self.act is not None:
|
||||
out1 = self.act(out1) + 1e-10
|
||||
out2 = self.act(out2) + 1e-10
|
||||
if self.use_log:
|
||||
# for recognition distillation, log is needed for feature map
|
||||
log_out1 = paddle.log(out1)
|
||||
log_out2 = paddle.log(out2)
|
||||
loss = (self._kldiv(log_out1, out2) + self._kldiv(log_out2, out1)) / 2.0
|
||||
else:
|
||||
# for detection distillation log is not needed
|
||||
loss = self.jskl_loss(out1, out2)
|
||||
return loss
|
||||
|
||||
|
||||
class DistanceLoss(nn.Layer):
|
||||
"""
|
||||
DistanceLoss:
|
||||
mode: loss mode
|
||||
"""
|
||||
|
||||
def __init__(self, mode="l2", **kargs):
|
||||
super().__init__()
|
||||
assert mode in ["l1", "l2", "smooth_l1"]
|
||||
if mode == "l1":
|
||||
self.loss_func = nn.L1Loss(**kargs)
|
||||
elif mode == "l2":
|
||||
self.loss_func = nn.MSELoss(**kargs)
|
||||
elif mode == "smooth_l1":
|
||||
self.loss_func = nn.SmoothL1Loss(**kargs)
|
||||
|
||||
def forward(self, x, y):
|
||||
return self.loss_func(x, y)
|
||||
|
||||
|
||||
class LossFromOutput(nn.Layer):
|
||||
def __init__(self, key="loss", reduction="none"):
|
||||
super().__init__()
|
||||
self.key = key
|
||||
self.reduction = reduction
|
||||
|
||||
def forward(self, predicts, batch):
|
||||
loss = predicts
|
||||
if self.key is not None and isinstance(predicts, dict):
|
||||
loss = loss[self.key]
|
||||
if self.reduction == "mean":
|
||||
loss = paddle.mean(loss)
|
||||
elif self.reduction == "sum":
|
||||
loss = paddle.sum(loss)
|
||||
return {"loss": loss}
|
||||
|
||||
|
||||
class KLDivLoss(nn.Layer):
|
||||
"""
|
||||
KLDivLoss
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
|
||||
def _kldiv(self, x, target, mask=None):
|
||||
eps = 1.0e-10
|
||||
loss = target * (paddle.log(target + eps) - x)
|
||||
if mask is not None:
|
||||
loss = loss.flatten(0, 1).sum(axis=1)
|
||||
loss = loss.masked_select(mask).mean()
|
||||
else:
|
||||
# batch mean loss
|
||||
loss = paddle.sum(loss) / loss.shape[0]
|
||||
return loss
|
||||
|
||||
def forward(self, logits_s, logits_t, mask=None):
|
||||
log_out_s = F.log_softmax(logits_s, axis=-1)
|
||||
out_t = F.softmax(logits_t, axis=-1)
|
||||
loss = self._kldiv(log_out_s, out_t, mask)
|
||||
return loss
|
||||
|
||||
|
||||
class DKDLoss(nn.Layer):
|
||||
"""
|
||||
KLDivLoss
|
||||
"""
|
||||
|
||||
def __init__(self, temperature=1.0, alpha=1.0, beta=1.0):
|
||||
super().__init__()
|
||||
self.temperature = temperature
|
||||
self.alpha = alpha
|
||||
self.beta = beta
|
||||
|
||||
def _cat_mask(self, t, mask1, mask2):
|
||||
t1 = (t * mask1).sum(axis=1, keepdim=True)
|
||||
t2 = (t * mask2).sum(axis=1, keepdim=True)
|
||||
rt = paddle.concat([t1, t2], axis=1)
|
||||
return rt
|
||||
|
||||
def _kl_div(self, x, label, mask=None):
|
||||
y = (label * (paddle.log(label + 1e-10) - x)).sum(axis=1)
|
||||
if mask is not None:
|
||||
y = y.masked_select(mask).mean()
|
||||
else:
|
||||
y = y.mean()
|
||||
return y
|
||||
|
||||
def forward(self, logits_student, logits_teacher, target, mask=None):
|
||||
gt_mask = F.one_hot(target.reshape([-1]), num_classes=logits_student.shape[-1])
|
||||
other_mask = 1 - gt_mask
|
||||
logits_student = logits_student.flatten(0, 1)
|
||||
logits_teacher = logits_teacher.flatten(0, 1)
|
||||
pred_student = F.softmax(logits_student / self.temperature, axis=1)
|
||||
pred_teacher = F.softmax(logits_teacher / self.temperature, axis=1)
|
||||
pred_student = self._cat_mask(pred_student, gt_mask, other_mask)
|
||||
pred_teacher = self._cat_mask(pred_teacher, gt_mask, other_mask)
|
||||
log_pred_student = paddle.log(pred_student)
|
||||
tckd_loss = self._kl_div(log_pred_student, pred_teacher) * (self.temperature**2)
|
||||
pred_teacher_part2 = F.softmax(
|
||||
logits_teacher / self.temperature - 1000.0 * gt_mask, axis=1
|
||||
)
|
||||
log_pred_student_part2 = F.log_softmax(
|
||||
logits_student / self.temperature - 1000.0 * gt_mask, axis=1
|
||||
)
|
||||
nckd_loss = self._kl_div(log_pred_student_part2, pred_teacher_part2) * (
|
||||
self.temperature**2
|
||||
)
|
||||
|
||||
loss = self.alpha * tckd_loss + self.beta * nckd_loss
|
||||
|
||||
return loss
|
||||
89
ppocr/losses/center_loss.py
Normal file
89
ppocr/losses/center_loss.py
Normal file
@@ -0,0 +1,89 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
# This code is refer from: https://github.com/KaiyangZhou/pytorch-center-loss
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
import os
|
||||
import pickle
|
||||
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
|
||||
|
||||
class CenterLoss(nn.Layer):
|
||||
"""
|
||||
Reference: Wen et al. A Discriminative Feature Learning Approach for Deep Face Recognition. ECCV 2016.
|
||||
"""
|
||||
|
||||
def __init__(self, num_classes=6625, feat_dim=96, center_file_path=None):
|
||||
super().__init__()
|
||||
self.num_classes = num_classes
|
||||
self.feat_dim = feat_dim
|
||||
self.centers = paddle.randn(shape=[self.num_classes, self.feat_dim]).astype(
|
||||
"float64"
|
||||
)
|
||||
|
||||
if center_file_path is not None:
|
||||
assert os.path.exists(
|
||||
center_file_path
|
||||
), f"center path({center_file_path}) must exist when it is not None."
|
||||
with open(center_file_path, "rb") as f:
|
||||
char_dict = pickle.load(f)
|
||||
for key in char_dict.keys():
|
||||
self.centers[key] = paddle.to_tensor(char_dict[key])
|
||||
|
||||
def __call__(self, predicts, batch):
|
||||
assert isinstance(predicts, (list, tuple))
|
||||
features, predicts = predicts
|
||||
|
||||
feats_reshape = paddle.reshape(features, [-1, features.shape[-1]]).astype(
|
||||
"float64"
|
||||
)
|
||||
label = paddle.argmax(predicts, axis=2)
|
||||
label = paddle.reshape(label, [label.shape[0] * label.shape[1]])
|
||||
|
||||
batch_size = feats_reshape.shape[0]
|
||||
|
||||
# calc l2 distance between feats and centers
|
||||
square_feat = paddle.sum(paddle.square(feats_reshape), axis=1, keepdim=True)
|
||||
square_feat = paddle.expand(square_feat, [batch_size, self.num_classes])
|
||||
|
||||
square_center = paddle.sum(paddle.square(self.centers), axis=1, keepdim=True)
|
||||
square_center = paddle.expand(
|
||||
square_center, [self.num_classes, batch_size]
|
||||
).astype("float64")
|
||||
square_center = paddle.transpose(square_center, [1, 0])
|
||||
|
||||
distmat = paddle.add(square_feat, square_center)
|
||||
feat_dot_center = paddle.matmul(
|
||||
feats_reshape, paddle.transpose(self.centers, [1, 0])
|
||||
)
|
||||
distmat = distmat - 2.0 * feat_dot_center
|
||||
|
||||
# generate the mask
|
||||
classes = paddle.arange(self.num_classes).astype("int64")
|
||||
label = paddle.expand(
|
||||
paddle.unsqueeze(label, 1), (batch_size, self.num_classes)
|
||||
)
|
||||
mask = paddle.equal(
|
||||
paddle.expand(classes, [batch_size, self.num_classes]), label
|
||||
).astype("float64")
|
||||
dist = paddle.multiply(distmat, mask)
|
||||
|
||||
loss = paddle.sum(paddle.clip(dist, min=1e-12, max=1e12)) / batch_size
|
||||
return {"loss_center": loss}
|
||||
30
ppocr/losses/cls_loss.py
Executable file
30
ppocr/losses/cls_loss.py
Executable file
@@ -0,0 +1,30 @@
|
||||
# copyright (c) 2019 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
from paddle import nn
|
||||
|
||||
|
||||
class ClsLoss(nn.Layer):
|
||||
def __init__(self, **kwargs):
|
||||
super(ClsLoss, self).__init__()
|
||||
self.loss_func = nn.CrossEntropyLoss(reduction="mean")
|
||||
|
||||
def forward(self, predicts, batch):
|
||||
label = batch[1].astype("int64")
|
||||
loss = self.loss_func(input=predicts, label=label)
|
||||
return {"loss": loss}
|
||||
84
ppocr/losses/combined_loss.py
Normal file
84
ppocr/losses/combined_loss.py
Normal file
@@ -0,0 +1,84 @@
|
||||
# Copyright (c) 2020 PaddlePaddle Authors. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
|
||||
from .rec_ctc_loss import CTCLoss
|
||||
from .center_loss import CenterLoss
|
||||
from .ace_loss import ACELoss
|
||||
from .rec_sar_loss import SARLoss
|
||||
|
||||
from .distillation_loss import DistillationCTCLoss, DistillCTCLogits
|
||||
from .distillation_loss import DistillationSARLoss, DistillationNRTRLoss
|
||||
from .distillation_loss import (
|
||||
DistillationDMLLoss,
|
||||
DistillationKLDivLoss,
|
||||
DistillationDKDLoss,
|
||||
)
|
||||
from .distillation_loss import (
|
||||
DistillationDistanceLoss,
|
||||
DistillationDBLoss,
|
||||
DistillationDilaDBLoss,
|
||||
)
|
||||
from .distillation_loss import (
|
||||
DistillationVQASerTokenLayoutLMLoss,
|
||||
DistillationSERDMLLoss,
|
||||
)
|
||||
from .distillation_loss import DistillationLossFromOutput
|
||||
from .distillation_loss import DistillationVQADistanceLoss
|
||||
|
||||
|
||||
class CombinedLoss(nn.Layer):
|
||||
"""
|
||||
CombinedLoss:
|
||||
a combionation of loss function
|
||||
"""
|
||||
|
||||
def __init__(self, loss_config_list=None):
|
||||
super().__init__()
|
||||
self.loss_func = []
|
||||
self.loss_weight = []
|
||||
assert isinstance(loss_config_list, list), "operator config should be a list"
|
||||
for config in loss_config_list:
|
||||
assert isinstance(config, dict) and len(config) == 1, "yaml format error"
|
||||
name = list(config)[0]
|
||||
param = config[name]
|
||||
assert (
|
||||
"weight" in param
|
||||
), "weight must be in param, but param just contains {}".format(
|
||||
param.keys()
|
||||
)
|
||||
self.loss_weight.append(param.pop("weight"))
|
||||
self.loss_func.append(eval(name)(**param))
|
||||
|
||||
def forward(self, input, batch, **kargs):
|
||||
loss_dict = {}
|
||||
loss_all = 0.0
|
||||
for idx, loss_func in enumerate(self.loss_func):
|
||||
loss = loss_func(input, batch, **kargs)
|
||||
if isinstance(loss, paddle.Tensor):
|
||||
loss = {"loss_{}_{}".format(str(loss), idx): loss}
|
||||
|
||||
weight = self.loss_weight[idx]
|
||||
|
||||
loss = {key: loss[key] * weight for key in loss}
|
||||
|
||||
if "loss" in loss:
|
||||
loss_all += loss["loss"]
|
||||
else:
|
||||
loss_all += paddle.add_n(list(loss.values()))
|
||||
loss_dict.update(loss)
|
||||
loss_dict["loss"] = loss_all
|
||||
return loss_dict
|
||||
161
ppocr/losses/det_basic_loss.py
Normal file
161
ppocr/losses/det_basic_loss.py
Normal file
@@ -0,0 +1,161 @@
|
||||
# copyright (c) 2019 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/WenmuZhou/DBNet.pytorch/blob/master/models/losses/basic_loss.py
|
||||
"""
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import numpy as np
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
import paddle.nn.functional as F
|
||||
|
||||
|
||||
class BalanceLoss(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
balance_loss=True,
|
||||
main_loss_type="DiceLoss",
|
||||
negative_ratio=3,
|
||||
return_origin=False,
|
||||
eps=1e-6,
|
||||
**kwargs,
|
||||
):
|
||||
"""
|
||||
The BalanceLoss for Differentiable Binarization text detection
|
||||
args:
|
||||
balance_loss (bool): whether balance loss or not, default is True
|
||||
main_loss_type (str): can only be one of ['CrossEntropy','DiceLoss',
|
||||
'Euclidean','BCELoss', 'MaskL1Loss'], default is 'DiceLoss'.
|
||||
negative_ratio (int|float): float, default is 3.
|
||||
return_origin (bool): whether return unbalanced loss or not, default is False.
|
||||
eps (float): default is 1e-6.
|
||||
"""
|
||||
super(BalanceLoss, self).__init__()
|
||||
self.balance_loss = balance_loss
|
||||
self.main_loss_type = main_loss_type
|
||||
self.negative_ratio = negative_ratio
|
||||
self.return_origin = return_origin
|
||||
self.eps = eps
|
||||
|
||||
if self.main_loss_type == "CrossEntropy":
|
||||
self.loss = nn.CrossEntropyLoss()
|
||||
elif self.main_loss_type == "Euclidean":
|
||||
self.loss = nn.MSELoss()
|
||||
elif self.main_loss_type == "DiceLoss":
|
||||
self.loss = DiceLoss(self.eps)
|
||||
elif self.main_loss_type == "BCELoss":
|
||||
self.loss = BCELoss(reduction="none")
|
||||
elif self.main_loss_type == "MaskL1Loss":
|
||||
self.loss = MaskL1Loss(self.eps)
|
||||
else:
|
||||
loss_type = [
|
||||
"CrossEntropy",
|
||||
"DiceLoss",
|
||||
"Euclidean",
|
||||
"BCELoss",
|
||||
"MaskL1Loss",
|
||||
]
|
||||
raise Exception(
|
||||
"main_loss_type in BalanceLoss() can only be one of {}".format(
|
||||
loss_type
|
||||
)
|
||||
)
|
||||
|
||||
def forward(self, pred, gt, mask=None):
|
||||
"""
|
||||
The BalanceLoss for Differentiable Binarization text detection
|
||||
args:
|
||||
pred (variable): predicted feature maps.
|
||||
gt (variable): ground truth feature maps.
|
||||
mask (variable): masked maps.
|
||||
return: (variable) balanced loss
|
||||
"""
|
||||
positive = gt * mask
|
||||
negative = (1 - gt) * mask
|
||||
|
||||
positive_count = int(positive.sum())
|
||||
negative_count = int(min(negative.sum(), positive_count * self.negative_ratio))
|
||||
loss = self.loss(pred, gt, mask=mask)
|
||||
|
||||
if not self.balance_loss:
|
||||
return loss
|
||||
|
||||
positive_loss = positive * loss
|
||||
negative_loss = negative * loss
|
||||
negative_loss = paddle.reshape(negative_loss, shape=[-1])
|
||||
if negative_count > 0:
|
||||
sort_loss = negative_loss.sort(descending=True)
|
||||
negative_loss = sort_loss[:negative_count]
|
||||
# negative_loss, _ = paddle.topk(negative_loss, k=negative_count_int)
|
||||
balance_loss = (positive_loss.sum() + negative_loss.sum()) / (
|
||||
positive_count + negative_count + self.eps
|
||||
)
|
||||
else:
|
||||
balance_loss = positive_loss.sum() / (positive_count + self.eps)
|
||||
if self.return_origin:
|
||||
return balance_loss, loss
|
||||
|
||||
return balance_loss
|
||||
|
||||
|
||||
class DiceLoss(nn.Layer):
|
||||
def __init__(self, eps=1e-6):
|
||||
super(DiceLoss, self).__init__()
|
||||
self.eps = eps
|
||||
|
||||
def forward(self, pred, gt, mask, weights=None):
|
||||
"""
|
||||
DiceLoss function.
|
||||
"""
|
||||
|
||||
assert pred.shape == gt.shape
|
||||
assert pred.shape == mask.shape
|
||||
if weights is not None:
|
||||
assert weights.shape == mask.shape
|
||||
mask = weights * mask
|
||||
intersection = paddle.sum(pred * gt * mask)
|
||||
|
||||
union = paddle.sum(pred * mask) + paddle.sum(gt * mask) + self.eps
|
||||
loss = 1 - 2.0 * intersection / union
|
||||
assert loss <= 1
|
||||
return loss
|
||||
|
||||
|
||||
class MaskL1Loss(nn.Layer):
|
||||
def __init__(self, eps=1e-6):
|
||||
super(MaskL1Loss, self).__init__()
|
||||
self.eps = eps
|
||||
|
||||
def forward(self, pred, gt, mask):
|
||||
"""
|
||||
Mask L1 Loss
|
||||
"""
|
||||
loss = (paddle.abs(pred - gt) * mask).sum() / (mask.sum() + self.eps)
|
||||
loss = paddle.mean(loss)
|
||||
return loss
|
||||
|
||||
|
||||
class BCELoss(nn.Layer):
|
||||
def __init__(self, reduction="mean"):
|
||||
super(BCELoss, self).__init__()
|
||||
self.reduction = reduction
|
||||
|
||||
def forward(self, input, label, mask=None, weight=None, name=None):
|
||||
loss = F.binary_cross_entropy(input, label, reduction=self.reduction)
|
||||
return loss
|
||||
302
ppocr/losses/det_ct_loss.py
Executable file
302
ppocr/losses/det_ct_loss.py
Executable file
@@ -0,0 +1,302 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/shengtao96/CentripetalText/tree/main/models/loss
|
||||
"""
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
import paddle.nn.functional as F
|
||||
import numpy as np
|
||||
|
||||
|
||||
def ohem_single(score, gt_text, training_mask):
|
||||
# online hard example mining
|
||||
|
||||
pos_num = int(paddle.sum(gt_text > 0.5)) - int(
|
||||
paddle.sum((gt_text > 0.5) & (training_mask <= 0.5))
|
||||
)
|
||||
|
||||
if pos_num == 0:
|
||||
# selected_mask = gt_text.copy() * 0 # may be not good
|
||||
selected_mask = training_mask
|
||||
selected_mask = paddle.cast(
|
||||
selected_mask.reshape((1, selected_mask.shape[0], selected_mask.shape[1])),
|
||||
"float32",
|
||||
)
|
||||
return selected_mask
|
||||
|
||||
neg_num = int(paddle.sum((gt_text <= 0.5) & (training_mask > 0.5)))
|
||||
neg_num = int(min(pos_num * 3, neg_num))
|
||||
|
||||
if neg_num == 0:
|
||||
selected_mask = training_mask
|
||||
selected_mask = paddle.cast(
|
||||
selected_mask.reshape((1, selected_mask.shape[0], selected_mask.shape[1])),
|
||||
"float32",
|
||||
)
|
||||
return selected_mask
|
||||
|
||||
# hard example
|
||||
neg_score = score[(gt_text <= 0.5) & (training_mask > 0.5)]
|
||||
neg_score_sorted = paddle.sort(-neg_score)
|
||||
threshold = -neg_score_sorted[neg_num - 1]
|
||||
|
||||
selected_mask = ((score >= threshold) | (gt_text > 0.5)) & (training_mask > 0.5)
|
||||
selected_mask = paddle.cast(
|
||||
selected_mask.reshape((1, selected_mask.shape[0], selected_mask.shape[1])),
|
||||
"float32",
|
||||
)
|
||||
return selected_mask
|
||||
|
||||
|
||||
def ohem_batch(scores, gt_texts, training_masks):
|
||||
selected_masks = []
|
||||
for i in range(scores.shape[0]):
|
||||
selected_masks.append(
|
||||
ohem_single(scores[i, :, :], gt_texts[i, :, :], training_masks[i, :, :])
|
||||
)
|
||||
|
||||
selected_masks = paddle.cast(paddle.concat(selected_masks, 0), "float32")
|
||||
return selected_masks
|
||||
|
||||
|
||||
def iou_single(a, b, mask, n_class):
|
||||
EPS = 1e-6
|
||||
valid = mask == 1
|
||||
a = a[valid]
|
||||
b = b[valid]
|
||||
miou = []
|
||||
|
||||
# iou of each class
|
||||
for i in range(n_class):
|
||||
inter = paddle.cast(((a == i) & (b == i)), "float32")
|
||||
union = paddle.cast(((a == i) | (b == i)), "float32")
|
||||
|
||||
miou.append(paddle.sum(inter) / (paddle.sum(union) + EPS))
|
||||
miou = sum(miou) / len(miou)
|
||||
return miou
|
||||
|
||||
|
||||
def iou(a, b, mask, n_class=2, reduce=True):
|
||||
batch_size = a.shape[0]
|
||||
|
||||
a = a.reshape((batch_size, -1))
|
||||
b = b.reshape((batch_size, -1))
|
||||
mask = mask.reshape((batch_size, -1))
|
||||
|
||||
iou = paddle.zeros((batch_size,), dtype="float32")
|
||||
for i in range(batch_size):
|
||||
iou[i] = iou_single(a[i], b[i], mask[i], n_class)
|
||||
|
||||
if reduce:
|
||||
iou = paddle.mean(iou)
|
||||
return iou
|
||||
|
||||
|
||||
class DiceLoss(nn.Layer):
|
||||
def __init__(self, loss_weight=1.0):
|
||||
super(DiceLoss, self).__init__()
|
||||
self.loss_weight = loss_weight
|
||||
|
||||
def forward(self, input, target, mask, reduce=True):
|
||||
batch_size = input.shape[0]
|
||||
input = F.sigmoid(input) # scale to 0-1
|
||||
|
||||
input = input.reshape((batch_size, -1))
|
||||
target = paddle.cast(target.reshape((batch_size, -1)), "float32")
|
||||
mask = paddle.cast(mask.reshape((batch_size, -1)), "float32")
|
||||
|
||||
input = input * mask
|
||||
target = target * mask
|
||||
|
||||
a = paddle.sum(input * target, axis=1)
|
||||
b = paddle.sum(input * input, axis=1) + 0.001
|
||||
c = paddle.sum(target * target, axis=1) + 0.001
|
||||
d = (2 * a) / (b + c)
|
||||
loss = 1 - d
|
||||
|
||||
loss = self.loss_weight * loss
|
||||
|
||||
if reduce:
|
||||
loss = paddle.mean(loss)
|
||||
|
||||
return loss
|
||||
|
||||
|
||||
class SmoothL1Loss(nn.Layer):
|
||||
def __init__(self, beta=1.0, loss_weight=1.0):
|
||||
super(SmoothL1Loss, self).__init__()
|
||||
self.beta = beta
|
||||
self.loss_weight = loss_weight
|
||||
|
||||
np_coord = np.zeros(shape=[640, 640, 2], dtype=np.int64)
|
||||
for i in range(640):
|
||||
for j in range(640):
|
||||
np_coord[i, j, 0] = j
|
||||
np_coord[i, j, 1] = i
|
||||
np_coord = np_coord.reshape((-1, 2))
|
||||
|
||||
self.coord = self.create_parameter(
|
||||
shape=[640 * 640, 2],
|
||||
dtype="int32", # NOTE: not support "int64" before paddle 2.3.1
|
||||
default_initializer=nn.initializer.Assign(value=np_coord),
|
||||
)
|
||||
self.coord.stop_gradient = True
|
||||
|
||||
def forward_single(self, input, target, mask, beta=1.0, eps=1e-6):
|
||||
batch_size = input.shape[0]
|
||||
|
||||
diff = paddle.abs(input - target) * mask.unsqueeze(1)
|
||||
loss = paddle.where(diff < beta, 0.5 * diff * diff / beta, diff - 0.5 * beta)
|
||||
loss = paddle.cast(loss.reshape((batch_size, -1)), "float32")
|
||||
mask = paddle.cast(mask.reshape((batch_size, -1)), "float32")
|
||||
loss = paddle.sum(loss, axis=-1)
|
||||
loss = loss / (mask.sum(axis=-1) + eps)
|
||||
|
||||
return loss
|
||||
|
||||
def select_single(self, distance, gt_instance, gt_kernel_instance, training_mask):
|
||||
with paddle.no_grad():
|
||||
# paddle 2.3.1, paddle.slice not support:
|
||||
# distance[:, self.coord[:, 1], self.coord[:, 0]]
|
||||
select_distance_list = []
|
||||
for i in range(2):
|
||||
tmp1 = distance[i, :]
|
||||
tmp2 = tmp1[self.coord[:, 1], self.coord[:, 0]]
|
||||
select_distance_list.append(tmp2.unsqueeze(0))
|
||||
select_distance = paddle.concat(select_distance_list, axis=0)
|
||||
|
||||
off_points = paddle.cast(
|
||||
self.coord, "float32"
|
||||
) + 10 * select_distance.transpose((1, 0))
|
||||
|
||||
off_points = paddle.cast(off_points, "int64")
|
||||
off_points = paddle.clip(off_points, 0, distance.shape[-1] - 1)
|
||||
|
||||
selected_mask = (
|
||||
gt_instance[self.coord[:, 1], self.coord[:, 0]]
|
||||
!= gt_kernel_instance[off_points[:, 1], off_points[:, 0]]
|
||||
)
|
||||
selected_mask = paddle.cast(
|
||||
selected_mask.reshape((1, -1, distance.shape[-1])), "int64"
|
||||
)
|
||||
selected_training_mask = selected_mask * training_mask
|
||||
|
||||
return selected_training_mask
|
||||
|
||||
def forward(
|
||||
self,
|
||||
distances,
|
||||
gt_instances,
|
||||
gt_kernel_instances,
|
||||
training_masks,
|
||||
gt_distances,
|
||||
reduce=True,
|
||||
):
|
||||
selected_training_masks = []
|
||||
for i in range(distances.shape[0]):
|
||||
selected_training_masks.append(
|
||||
self.select_single(
|
||||
distances[i, :, :, :],
|
||||
gt_instances[i, :, :],
|
||||
gt_kernel_instances[i, :, :],
|
||||
training_masks[i, :, :],
|
||||
)
|
||||
)
|
||||
selected_training_masks = paddle.cast(
|
||||
paddle.concat(selected_training_masks, 0), "float32"
|
||||
)
|
||||
|
||||
loss = self.forward_single(
|
||||
distances, gt_distances, selected_training_masks, self.beta
|
||||
)
|
||||
loss = self.loss_weight * loss
|
||||
|
||||
with paddle.no_grad():
|
||||
batch_size = distances.shape[0]
|
||||
false_num = selected_training_masks.reshape((batch_size, -1))
|
||||
false_num = false_num.sum(axis=-1)
|
||||
total_num = paddle.cast(training_masks.reshape((batch_size, -1)), "float32")
|
||||
total_num = total_num.sum(axis=-1)
|
||||
iou_text = (total_num - false_num) / (total_num + 1e-6)
|
||||
|
||||
if reduce:
|
||||
loss = paddle.mean(loss)
|
||||
|
||||
return loss, iou_text
|
||||
|
||||
|
||||
class CTLoss(nn.Layer):
|
||||
def __init__(self):
|
||||
super(CTLoss, self).__init__()
|
||||
self.kernel_loss = DiceLoss()
|
||||
self.loc_loss = SmoothL1Loss(beta=0.1, loss_weight=0.05)
|
||||
|
||||
def forward(self, preds, batch):
|
||||
imgs = batch[0]
|
||||
out = preds["maps"]
|
||||
(
|
||||
gt_kernels,
|
||||
training_masks,
|
||||
gt_instances,
|
||||
gt_kernel_instances,
|
||||
training_mask_distances,
|
||||
gt_distances,
|
||||
) = batch[1:]
|
||||
|
||||
kernels = out[:, 0, :, :]
|
||||
distances = out[:, 1:, :, :]
|
||||
|
||||
# kernel loss
|
||||
selected_masks = ohem_batch(kernels, gt_kernels, training_masks)
|
||||
|
||||
loss_kernel = self.kernel_loss(
|
||||
kernels, gt_kernels, selected_masks, reduce=False
|
||||
)
|
||||
|
||||
iou_kernel = iou(
|
||||
paddle.cast((kernels > 0), "int64"),
|
||||
gt_kernels,
|
||||
training_masks,
|
||||
reduce=False,
|
||||
)
|
||||
losses = dict(
|
||||
loss_kernels=loss_kernel,
|
||||
)
|
||||
|
||||
# loc loss
|
||||
loss_loc, iou_text = self.loc_loss(
|
||||
distances,
|
||||
gt_instances,
|
||||
gt_kernel_instances,
|
||||
training_mask_distances,
|
||||
gt_distances,
|
||||
reduce=False,
|
||||
)
|
||||
losses.update(
|
||||
dict(
|
||||
loss_loc=loss_loc,
|
||||
)
|
||||
)
|
||||
|
||||
loss_all = loss_kernel + loss_loc
|
||||
losses = {"loss": loss_all}
|
||||
|
||||
return losses
|
||||
99
ppocr/losses/det_db_loss.py
Executable file
99
ppocr/losses/det_db_loss.py
Executable file
@@ -0,0 +1,99 @@
|
||||
# copyright (c) 2019 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/WenmuZhou/DBNet.pytorch/blob/master/models/losses/DB_loss.py
|
||||
"""
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
|
||||
from .det_basic_loss import BalanceLoss, MaskL1Loss, DiceLoss
|
||||
|
||||
|
||||
class DBLoss(nn.Layer):
|
||||
"""
|
||||
Differentiable Binarization (DB) Loss Function
|
||||
args:
|
||||
param (dict): the super parameter for DB Loss
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
balance_loss=True,
|
||||
main_loss_type="DiceLoss",
|
||||
alpha=5,
|
||||
beta=10,
|
||||
ohem_ratio=3,
|
||||
eps=1e-6,
|
||||
**kwargs,
|
||||
):
|
||||
super(DBLoss, self).__init__()
|
||||
self.alpha = alpha
|
||||
self.beta = beta
|
||||
self.dice_loss = DiceLoss(eps=eps)
|
||||
self.l1_loss = MaskL1Loss(eps=eps)
|
||||
self.bce_loss = BalanceLoss(
|
||||
balance_loss=balance_loss,
|
||||
main_loss_type=main_loss_type,
|
||||
negative_ratio=ohem_ratio,
|
||||
)
|
||||
|
||||
def forward(self, predicts, labels):
|
||||
predict_maps = predicts["maps"]
|
||||
(
|
||||
label_threshold_map,
|
||||
label_threshold_mask,
|
||||
label_shrink_map,
|
||||
label_shrink_mask,
|
||||
) = labels[1:]
|
||||
shrink_maps = predict_maps[:, 0, :, :]
|
||||
threshold_maps = predict_maps[:, 1, :, :]
|
||||
binary_maps = predict_maps[:, 2, :, :]
|
||||
|
||||
loss_shrink_maps = self.bce_loss(
|
||||
shrink_maps, label_shrink_map, label_shrink_mask
|
||||
)
|
||||
loss_threshold_maps = self.l1_loss(
|
||||
threshold_maps, label_threshold_map, label_threshold_mask
|
||||
)
|
||||
loss_binary_maps = self.dice_loss(
|
||||
binary_maps, label_shrink_map, label_shrink_mask
|
||||
)
|
||||
loss_shrink_maps = self.alpha * loss_shrink_maps
|
||||
loss_threshold_maps = self.beta * loss_threshold_maps
|
||||
# CBN loss
|
||||
if "distance_maps" in predicts.keys():
|
||||
distance_maps = predicts["distance_maps"]
|
||||
cbn_maps = predicts["cbn_maps"]
|
||||
cbn_loss = self.bce_loss(
|
||||
cbn_maps[:, 0, :, :], label_shrink_map, label_shrink_mask
|
||||
)
|
||||
else:
|
||||
dis_loss = paddle.to_tensor([0.0])
|
||||
cbn_loss = paddle.to_tensor([0.0])
|
||||
|
||||
loss_all = loss_shrink_maps + loss_threshold_maps + loss_binary_maps
|
||||
losses = {
|
||||
"loss": loss_all + cbn_loss,
|
||||
"loss_shrink_maps": loss_shrink_maps,
|
||||
"loss_threshold_maps": loss_threshold_maps,
|
||||
"loss_binary_maps": loss_binary_maps,
|
||||
"loss_cbn": cbn_loss,
|
||||
}
|
||||
return losses
|
||||
234
ppocr/losses/det_drrg_loss.py
Normal file
234
ppocr/losses/det_drrg_loss.py
Normal file
@@ -0,0 +1,234 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/open-mmlab/mmocr/blob/main/mmocr/models/textdet/losses/drrg_loss.py
|
||||
"""
|
||||
|
||||
import paddle
|
||||
import paddle.nn.functional as F
|
||||
from paddle import nn
|
||||
|
||||
|
||||
class DRRGLoss(nn.Layer):
|
||||
def __init__(self, ohem_ratio=3.0):
|
||||
super().__init__()
|
||||
self.ohem_ratio = ohem_ratio
|
||||
self.downsample_ratio = 1.0
|
||||
|
||||
def balance_bce_loss(self, pred, gt, mask):
|
||||
"""Balanced Binary-CrossEntropy Loss.
|
||||
|
||||
Args:
|
||||
pred (Tensor): Shape of :math:`(1, H, W)`.
|
||||
gt (Tensor): Shape of :math:`(1, H, W)`.
|
||||
mask (Tensor): Shape of :math:`(1, H, W)`.
|
||||
|
||||
Returns:
|
||||
Tensor: Balanced bce loss.
|
||||
"""
|
||||
assert pred.shape == gt.shape == mask.shape
|
||||
assert paddle.all(pred >= 0) and paddle.all(pred <= 1)
|
||||
assert paddle.all(gt >= 0) and paddle.all(gt <= 1)
|
||||
positive = gt * mask
|
||||
negative = (1 - gt) * mask
|
||||
positive_count = int(positive.sum())
|
||||
|
||||
if positive_count > 0:
|
||||
loss = F.binary_cross_entropy(pred, gt, reduction="none")
|
||||
positive_loss = paddle.sum(loss * positive)
|
||||
negative_loss = loss * negative
|
||||
negative_count = min(
|
||||
int(negative.sum()), int(positive_count * self.ohem_ratio)
|
||||
)
|
||||
else:
|
||||
positive_loss = paddle.to_tensor(0.0)
|
||||
loss = F.binary_cross_entropy(pred, gt, reduction="none")
|
||||
negative_loss = loss * negative
|
||||
negative_count = 100
|
||||
negative_loss, _ = paddle.topk(negative_loss.reshape([-1]), negative_count)
|
||||
|
||||
balance_loss = (positive_loss + paddle.sum(negative_loss)) / (
|
||||
float(positive_count + negative_count) + 1e-5
|
||||
)
|
||||
|
||||
return balance_loss
|
||||
|
||||
def gcn_loss(self, gcn_data):
|
||||
"""CrossEntropy Loss from gcn module.
|
||||
|
||||
Args:
|
||||
gcn_data (tuple(Tensor, Tensor)): The first is the
|
||||
prediction with shape :math:`(N, 2)` and the
|
||||
second is the gt label with shape :math:`(m, n)`
|
||||
where :math:`m * n = N`.
|
||||
|
||||
Returns:
|
||||
Tensor: CrossEntropy loss.
|
||||
"""
|
||||
gcn_pred, gt_labels = gcn_data
|
||||
gt_labels = gt_labels.reshape([-1])
|
||||
loss = F.cross_entropy(gcn_pred, gt_labels)
|
||||
|
||||
return loss
|
||||
|
||||
def bitmasks2tensor(self, bitmasks, target_sz):
|
||||
"""Convert Bitmasks to tensor.
|
||||
|
||||
Args:
|
||||
bitmasks (list[BitmapMasks]): The BitmapMasks list. Each item is
|
||||
for one img.
|
||||
target_sz (tuple(int, int)): The target tensor of size
|
||||
:math:`(H, W)`.
|
||||
|
||||
Returns:
|
||||
list[Tensor]: The list of kernel tensors. Each element stands for
|
||||
one kernel level.
|
||||
"""
|
||||
batch_size = len(bitmasks)
|
||||
results = []
|
||||
|
||||
kernel = []
|
||||
for batch_inx in range(batch_size):
|
||||
mask = bitmasks[batch_inx]
|
||||
# hxw
|
||||
mask_sz = mask.shape
|
||||
# left, right, top, bottom
|
||||
pad = [0, target_sz[1] - mask_sz[1], 0, target_sz[0] - mask_sz[0]]
|
||||
mask = F.pad(mask, pad, mode="constant", value=0)
|
||||
kernel.append(mask)
|
||||
kernel = paddle.stack(kernel)
|
||||
results.append(kernel)
|
||||
|
||||
return results
|
||||
|
||||
def forward(self, preds, labels):
|
||||
"""Compute Drrg loss."""
|
||||
|
||||
assert isinstance(preds, tuple)
|
||||
(
|
||||
gt_text_mask,
|
||||
gt_center_region_mask,
|
||||
gt_mask,
|
||||
gt_top_height_map,
|
||||
gt_bot_height_map,
|
||||
gt_sin_map,
|
||||
gt_cos_map,
|
||||
) = labels[1:8]
|
||||
|
||||
downsample_ratio = self.downsample_ratio
|
||||
|
||||
pred_maps, gcn_data = preds
|
||||
pred_text_region = pred_maps[:, 0, :, :]
|
||||
pred_center_region = pred_maps[:, 1, :, :]
|
||||
pred_sin_map = pred_maps[:, 2, :, :]
|
||||
pred_cos_map = pred_maps[:, 3, :, :]
|
||||
pred_top_height_map = pred_maps[:, 4, :, :]
|
||||
pred_bot_height_map = pred_maps[:, 5, :, :]
|
||||
feature_sz = pred_maps.shape
|
||||
|
||||
# bitmask 2 tensor
|
||||
mapping = {
|
||||
"gt_text_mask": paddle.cast(gt_text_mask, "float32"),
|
||||
"gt_center_region_mask": paddle.cast(gt_center_region_mask, "float32"),
|
||||
"gt_mask": paddle.cast(gt_mask, "float32"),
|
||||
"gt_top_height_map": paddle.cast(gt_top_height_map, "float32"),
|
||||
"gt_bot_height_map": paddle.cast(gt_bot_height_map, "float32"),
|
||||
"gt_sin_map": paddle.cast(gt_sin_map, "float32"),
|
||||
"gt_cos_map": paddle.cast(gt_cos_map, "float32"),
|
||||
}
|
||||
gt = {}
|
||||
for key, value in mapping.items():
|
||||
gt[key] = value
|
||||
if abs(downsample_ratio - 1.0) < 1e-2:
|
||||
gt[key] = self.bitmasks2tensor(gt[key], feature_sz[2:])
|
||||
else:
|
||||
gt[key] = [item.rescale(downsample_ratio) for item in gt[key]]
|
||||
gt[key] = self.bitmasks2tensor(gt[key], feature_sz[2:])
|
||||
if key in ["gt_top_height_map", "gt_bot_height_map"]:
|
||||
gt[key] = [item * downsample_ratio for item in gt[key]]
|
||||
gt[key] = [item for item in gt[key]]
|
||||
|
||||
scale = paddle.sqrt(1.0 / (pred_sin_map**2 + pred_cos_map**2 + 1e-8))
|
||||
pred_sin_map = pred_sin_map * scale
|
||||
pred_cos_map = pred_cos_map * scale
|
||||
|
||||
loss_text = self.balance_bce_loss(
|
||||
F.sigmoid(pred_text_region), gt["gt_text_mask"][0], gt["gt_mask"][0]
|
||||
)
|
||||
|
||||
text_mask = gt["gt_text_mask"][0] * gt["gt_mask"][0]
|
||||
negative_text_mask = (1 - gt["gt_text_mask"][0]) * gt["gt_mask"][0]
|
||||
loss_center_map = F.binary_cross_entropy(
|
||||
F.sigmoid(pred_center_region),
|
||||
gt["gt_center_region_mask"][0],
|
||||
reduction="none",
|
||||
)
|
||||
if int(text_mask.sum()) > 0:
|
||||
loss_center_positive = paddle.sum(loss_center_map * text_mask) / paddle.sum(
|
||||
text_mask
|
||||
)
|
||||
else:
|
||||
loss_center_positive = paddle.to_tensor(0.0)
|
||||
loss_center_negative = paddle.sum(
|
||||
loss_center_map * negative_text_mask
|
||||
) / paddle.sum(negative_text_mask)
|
||||
loss_center = loss_center_positive + 0.5 * loss_center_negative
|
||||
|
||||
center_mask = gt["gt_center_region_mask"][0] * gt["gt_mask"][0]
|
||||
if int(center_mask.sum()) > 0:
|
||||
map_sz = pred_top_height_map.shape
|
||||
ones = paddle.ones(map_sz, dtype="float32")
|
||||
loss_top = F.smooth_l1_loss(
|
||||
pred_top_height_map / (gt["gt_top_height_map"][0] + 1e-2),
|
||||
ones,
|
||||
reduction="none",
|
||||
)
|
||||
loss_bot = F.smooth_l1_loss(
|
||||
pred_bot_height_map / (gt["gt_bot_height_map"][0] + 1e-2),
|
||||
ones,
|
||||
reduction="none",
|
||||
)
|
||||
gt_height = gt["gt_top_height_map"][0] + gt["gt_bot_height_map"][0]
|
||||
loss_height = paddle.sum(
|
||||
(paddle.log(gt_height + 1) * (loss_top + loss_bot)) * center_mask
|
||||
) / paddle.sum(center_mask)
|
||||
|
||||
loss_sin = paddle.sum(
|
||||
F.smooth_l1_loss(pred_sin_map, gt["gt_sin_map"][0], reduction="none")
|
||||
* center_mask
|
||||
) / paddle.sum(center_mask)
|
||||
loss_cos = paddle.sum(
|
||||
F.smooth_l1_loss(pred_cos_map, gt["gt_cos_map"][0], reduction="none")
|
||||
* center_mask
|
||||
) / paddle.sum(center_mask)
|
||||
else:
|
||||
loss_height = paddle.to_tensor(0.0)
|
||||
loss_sin = paddle.to_tensor(0.0)
|
||||
loss_cos = paddle.to_tensor(0.0)
|
||||
|
||||
loss_gcn = self.gcn_loss(gcn_data)
|
||||
|
||||
loss = loss_text + loss_center + loss_height + loss_sin + loss_cos + loss_gcn
|
||||
results = dict(
|
||||
loss=loss,
|
||||
loss_text=loss_text,
|
||||
loss_center=loss_center,
|
||||
loss_height=loss_height,
|
||||
loss_sin=loss_sin,
|
||||
loss_cos=loss_cos,
|
||||
loss_gcn=loss_gcn,
|
||||
)
|
||||
|
||||
return results
|
||||
62
ppocr/losses/det_east_loss.py
Normal file
62
ppocr/losses/det_east_loss.py
Normal file
@@ -0,0 +1,62 @@
|
||||
# copyright (c) 2019 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
from .det_basic_loss import DiceLoss
|
||||
|
||||
|
||||
class EASTLoss(nn.Layer):
|
||||
""" """
|
||||
|
||||
def __init__(self, eps=1e-6, **kwargs):
|
||||
super(EASTLoss, self).__init__()
|
||||
self.dice_loss = DiceLoss(eps=eps)
|
||||
|
||||
def forward(self, predicts, labels):
|
||||
l_score, l_geo, l_mask = labels[1:]
|
||||
f_score = predicts["f_score"]
|
||||
f_geo = predicts["f_geo"]
|
||||
|
||||
dice_loss = self.dice_loss(f_score, l_score, l_mask)
|
||||
|
||||
# smoooth_l1_loss
|
||||
channels = 8
|
||||
l_geo_split = paddle.split(l_geo, num_or_sections=channels + 1, axis=1)
|
||||
f_geo_split = paddle.split(f_geo, num_or_sections=channels, axis=1)
|
||||
smooth_l1 = 0
|
||||
for i in range(0, channels):
|
||||
geo_diff = l_geo_split[i] - f_geo_split[i]
|
||||
abs_geo_diff = paddle.abs(geo_diff)
|
||||
smooth_l1_sign = paddle.less_than(abs_geo_diff, l_score)
|
||||
smooth_l1_sign = paddle.cast(smooth_l1_sign, dtype="float32")
|
||||
in_loss = abs_geo_diff * abs_geo_diff * smooth_l1_sign + (
|
||||
abs_geo_diff - 0.5
|
||||
) * (1.0 - smooth_l1_sign)
|
||||
out_loss = l_geo_split[-1] / channels * in_loss * l_score
|
||||
smooth_l1 += out_loss
|
||||
smooth_l1_loss = paddle.mean(smooth_l1 * l_score)
|
||||
|
||||
dice_loss = dice_loss * 0.01
|
||||
total_loss = dice_loss + smooth_l1_loss
|
||||
losses = {
|
||||
"loss": total_loss,
|
||||
"dice_loss": dice_loss,
|
||||
"smooth_l1_loss": smooth_l1_loss,
|
||||
}
|
||||
return losses
|
||||
240
ppocr/losses/det_fce_loss.py
Normal file
240
ppocr/losses/det_fce_loss.py
Normal file
@@ -0,0 +1,240 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/open-mmlab/mmocr/blob/main/mmocr/models/textdet/losses/fce_loss.py
|
||||
"""
|
||||
|
||||
import numpy as np
|
||||
from paddle import nn
|
||||
import paddle
|
||||
import paddle.nn.functional as F
|
||||
from functools import partial
|
||||
|
||||
|
||||
def multi_apply(func, *args, **kwargs):
|
||||
pfunc = partial(func, **kwargs) if kwargs else func
|
||||
map_results = map(pfunc, *args)
|
||||
return tuple(map(list, zip(*map_results)))
|
||||
|
||||
|
||||
class FCELoss(nn.Layer):
|
||||
"""The class for implementing FCENet loss
|
||||
FCENet(CVPR2021): Fourier Contour Embedding for Arbitrary-shaped
|
||||
Text Detection
|
||||
|
||||
[https://arxiv.org/abs/2104.10442]
|
||||
|
||||
Args:
|
||||
fourier_degree (int) : The maximum Fourier transform degree k.
|
||||
num_sample (int) : The sampling points number of regression
|
||||
loss. If it is too small, fcenet tends to be overfitting.
|
||||
ohem_ratio (float): the negative/positive ratio in OHEM.
|
||||
"""
|
||||
|
||||
def __init__(self, fourier_degree, num_sample, ohem_ratio=3.0):
|
||||
super().__init__()
|
||||
self.fourier_degree = fourier_degree
|
||||
self.num_sample = num_sample
|
||||
self.ohem_ratio = ohem_ratio
|
||||
|
||||
def forward(self, preds, labels):
|
||||
assert isinstance(preds, dict)
|
||||
preds = preds["levels"]
|
||||
|
||||
p3_maps, p4_maps, p5_maps = labels[1:]
|
||||
assert (
|
||||
p3_maps[0].shape[0] == 4 * self.fourier_degree + 5
|
||||
), "fourier degree not equal in FCEhead and FCEtarget"
|
||||
|
||||
# to tensor
|
||||
gts = [p3_maps, p4_maps, p5_maps]
|
||||
for idx, maps in enumerate(gts):
|
||||
gts[idx] = paddle.to_tensor(np.stack(maps))
|
||||
|
||||
losses = multi_apply(self.forward_single, preds, gts)
|
||||
|
||||
loss_tr = paddle.to_tensor(0.0).astype("float32")
|
||||
loss_tcl = paddle.to_tensor(0.0).astype("float32")
|
||||
loss_reg_x = paddle.to_tensor(0.0).astype("float32")
|
||||
loss_reg_y = paddle.to_tensor(0.0).astype("float32")
|
||||
loss_all = paddle.to_tensor(0.0).astype("float32")
|
||||
|
||||
for idx, loss in enumerate(losses):
|
||||
loss_all += sum(loss)
|
||||
if idx == 0:
|
||||
loss_tr += sum(loss)
|
||||
elif idx == 1:
|
||||
loss_tcl += sum(loss)
|
||||
elif idx == 2:
|
||||
loss_reg_x += sum(loss)
|
||||
else:
|
||||
loss_reg_y += sum(loss)
|
||||
|
||||
results = dict(
|
||||
loss=loss_all,
|
||||
loss_text=loss_tr,
|
||||
loss_center=loss_tcl,
|
||||
loss_reg_x=loss_reg_x,
|
||||
loss_reg_y=loss_reg_y,
|
||||
)
|
||||
return results
|
||||
|
||||
def forward_single(self, pred, gt):
|
||||
cls_pred = paddle.transpose(pred[0], (0, 2, 3, 1))
|
||||
reg_pred = paddle.transpose(pred[1], (0, 2, 3, 1))
|
||||
gt = paddle.transpose(gt, (0, 2, 3, 1))
|
||||
|
||||
k = 2 * self.fourier_degree + 1
|
||||
tr_pred = paddle.reshape(cls_pred[:, :, :, :2], (-1, 2))
|
||||
tcl_pred = paddle.reshape(cls_pred[:, :, :, 2:], (-1, 2))
|
||||
x_pred = paddle.reshape(reg_pred[:, :, :, 0:k], (-1, k))
|
||||
y_pred = paddle.reshape(reg_pred[:, :, :, k : 2 * k], (-1, k))
|
||||
|
||||
tr_mask = gt[:, :, :, :1].reshape([-1])
|
||||
tcl_mask = gt[:, :, :, 1:2].reshape([-1])
|
||||
train_mask = gt[:, :, :, 2:3].reshape([-1])
|
||||
x_map = paddle.reshape(gt[:, :, :, 3 : 3 + k], (-1, k))
|
||||
y_map = paddle.reshape(gt[:, :, :, 3 + k :], (-1, k))
|
||||
|
||||
tr_train_mask = (train_mask * tr_mask).astype("bool")
|
||||
tr_train_mask2 = paddle.concat(
|
||||
[tr_train_mask.unsqueeze(1), tr_train_mask.unsqueeze(1)], axis=1
|
||||
)
|
||||
# tr loss
|
||||
loss_tr = self.ohem(tr_pred, tr_mask, train_mask)
|
||||
# tcl loss
|
||||
loss_tcl = paddle.to_tensor(0.0).astype("float32")
|
||||
tr_neg_mask = tr_train_mask.logical_not()
|
||||
tr_neg_mask2 = paddle.concat(
|
||||
[tr_neg_mask.unsqueeze(1), tr_neg_mask.unsqueeze(1)], axis=1
|
||||
)
|
||||
if tr_train_mask.sum().item() > 0:
|
||||
loss_tcl_pos = F.cross_entropy(
|
||||
tcl_pred.masked_select(tr_train_mask2).reshape([-1, 2]),
|
||||
tcl_mask.masked_select(tr_train_mask).astype("int64"),
|
||||
)
|
||||
loss_tcl_neg = F.cross_entropy(
|
||||
tcl_pred.masked_select(tr_neg_mask2).reshape([-1, 2]),
|
||||
tcl_mask.masked_select(tr_neg_mask).astype("int64"),
|
||||
)
|
||||
loss_tcl = loss_tcl_pos + 0.5 * loss_tcl_neg
|
||||
|
||||
# regression loss
|
||||
loss_reg_x = paddle.to_tensor(0.0).astype("float32")
|
||||
loss_reg_y = paddle.to_tensor(0.0).astype("float32")
|
||||
if tr_train_mask.sum().item() > 0:
|
||||
weight = (
|
||||
tr_mask.masked_select(tr_train_mask.astype("bool")).astype("float32")
|
||||
+ tcl_mask.masked_select(tr_train_mask.astype("bool")).astype("float32")
|
||||
) / 2
|
||||
weight = weight.reshape([-1, 1])
|
||||
|
||||
ft_x, ft_y = self.fourier2poly(x_map, y_map)
|
||||
ft_x_pre, ft_y_pre = self.fourier2poly(x_pred, y_pred)
|
||||
|
||||
dim = ft_x.shape[1]
|
||||
|
||||
tr_train_mask3 = paddle.concat(
|
||||
[tr_train_mask.unsqueeze(1) for i in range(dim)], axis=1
|
||||
)
|
||||
|
||||
loss_reg_x = paddle.mean(
|
||||
weight
|
||||
* F.smooth_l1_loss(
|
||||
ft_x_pre.masked_select(tr_train_mask3).reshape([-1, dim]),
|
||||
ft_x.masked_select(tr_train_mask3).reshape([-1, dim]),
|
||||
reduction="none",
|
||||
)
|
||||
)
|
||||
loss_reg_y = paddle.mean(
|
||||
weight
|
||||
* F.smooth_l1_loss(
|
||||
ft_y_pre.masked_select(tr_train_mask3).reshape([-1, dim]),
|
||||
ft_y.masked_select(tr_train_mask3).reshape([-1, dim]),
|
||||
reduction="none",
|
||||
)
|
||||
)
|
||||
|
||||
return loss_tr, loss_tcl, loss_reg_x, loss_reg_y
|
||||
|
||||
def ohem(self, predict, target, train_mask):
|
||||
pos = (target * train_mask).astype("bool")
|
||||
neg = ((1 - target) * train_mask).astype("bool")
|
||||
|
||||
pos2 = paddle.concat([pos.unsqueeze(1), pos.unsqueeze(1)], axis=1)
|
||||
neg2 = paddle.concat([neg.unsqueeze(1), neg.unsqueeze(1)], axis=1)
|
||||
|
||||
n_pos = pos.astype("float32").sum()
|
||||
|
||||
if n_pos.item() > 0:
|
||||
loss_pos = F.cross_entropy(
|
||||
predict.masked_select(pos2).reshape([-1, 2]),
|
||||
target.masked_select(pos).astype("int64"),
|
||||
reduction="sum",
|
||||
)
|
||||
loss_neg = F.cross_entropy(
|
||||
predict.masked_select(neg2).reshape([-1, 2]),
|
||||
target.masked_select(neg).astype("int64"),
|
||||
reduction="none",
|
||||
)
|
||||
n_neg = min(
|
||||
int(neg.astype("float32").sum().item()),
|
||||
int(self.ohem_ratio * n_pos.astype("float32")),
|
||||
)
|
||||
else:
|
||||
loss_pos = paddle.to_tensor(0.0)
|
||||
loss_neg = F.cross_entropy(
|
||||
predict.masked_select(neg2).reshape([-1, 2]),
|
||||
target.masked_select(neg).astype("int64"),
|
||||
reduction="none",
|
||||
)
|
||||
n_neg = 100
|
||||
if len(loss_neg) > n_neg:
|
||||
loss_neg, _ = paddle.topk(loss_neg, n_neg)
|
||||
|
||||
return (loss_pos + loss_neg.sum()) / (n_pos + n_neg).astype("float32")
|
||||
|
||||
def fourier2poly(self, real_maps, imag_maps):
|
||||
"""Transform Fourier coefficient maps to polygon maps.
|
||||
|
||||
Args:
|
||||
real_maps (tensor): A map composed of the real parts of the
|
||||
Fourier coefficients, whose shape is (-1, 2k+1)
|
||||
imag_maps (tensor):A map composed of the imag parts of the
|
||||
Fourier coefficients, whose shape is (-1, 2k+1)
|
||||
|
||||
Returns
|
||||
x_maps (tensor): A map composed of the x value of the polygon
|
||||
represented by n sample points (xn, yn), whose shape is (-1, n)
|
||||
y_maps (tensor): A map composed of the y value of the polygon
|
||||
represented by n sample points (xn, yn), whose shape is (-1, n)
|
||||
"""
|
||||
|
||||
k_vect = paddle.arange(
|
||||
-self.fourier_degree, self.fourier_degree + 1, dtype="float32"
|
||||
).reshape([-1, 1])
|
||||
i_vect = paddle.arange(0, self.num_sample, dtype="float32").reshape([1, -1])
|
||||
|
||||
transform_matrix = 2 * np.pi / self.num_sample * paddle.matmul(k_vect, i_vect)
|
||||
|
||||
x1 = paddle.einsum("ak, kn-> an", real_maps, paddle.cos(transform_matrix))
|
||||
x2 = paddle.einsum("ak, kn-> an", imag_maps, paddle.sin(transform_matrix))
|
||||
y1 = paddle.einsum("ak, kn-> an", real_maps, paddle.sin(transform_matrix))
|
||||
y2 = paddle.einsum("ak, kn-> an", imag_maps, paddle.cos(transform_matrix))
|
||||
|
||||
x_maps = x1 - x2
|
||||
y_maps = y1 + y2
|
||||
|
||||
return x_maps, y_maps
|
||||
158
ppocr/losses/det_pse_loss.py
Normal file
158
ppocr/losses/det_pse_loss.py
Normal file
@@ -0,0 +1,158 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/whai362/PSENet/blob/python3/models/head/psenet_head.py
|
||||
"""
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
from paddle.nn import functional as F
|
||||
import numpy as np
|
||||
from ppocr.utils.iou import iou
|
||||
|
||||
|
||||
class PSELoss(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
alpha,
|
||||
ohem_ratio=3,
|
||||
kernel_sample_mask="pred",
|
||||
reduction="sum",
|
||||
eps=1e-6,
|
||||
**kwargs,
|
||||
):
|
||||
"""Implement PSE Loss."""
|
||||
super(PSELoss, self).__init__()
|
||||
assert reduction in ["sum", "mean", "none"]
|
||||
self.alpha = alpha
|
||||
self.ohem_ratio = ohem_ratio
|
||||
self.kernel_sample_mask = kernel_sample_mask
|
||||
self.reduction = reduction
|
||||
self.eps = eps
|
||||
|
||||
def forward(self, outputs, labels):
|
||||
predicts = outputs["maps"]
|
||||
predicts = F.interpolate(predicts, scale_factor=4)
|
||||
|
||||
texts = predicts[:, 0, :, :]
|
||||
kernels = predicts[:, 1:, :, :]
|
||||
gt_texts, gt_kernels, training_masks = labels[1:]
|
||||
|
||||
# text loss
|
||||
selected_masks = self.ohem_batch(texts, gt_texts, training_masks)
|
||||
|
||||
loss_text = self.dice_loss(texts, gt_texts, selected_masks)
|
||||
iou_text = iou(
|
||||
(texts > 0).astype("int64"), gt_texts, training_masks, reduce=False
|
||||
)
|
||||
losses = dict(loss_text=loss_text, iou_text=iou_text)
|
||||
|
||||
# kernel loss
|
||||
loss_kernels = []
|
||||
if self.kernel_sample_mask == "gt":
|
||||
selected_masks = gt_texts * training_masks
|
||||
elif self.kernel_sample_mask == "pred":
|
||||
selected_masks = (F.sigmoid(texts) > 0.5).astype("float32") * training_masks
|
||||
|
||||
for i in range(kernels.shape[1]):
|
||||
kernel_i = kernels[:, i, :, :]
|
||||
gt_kernel_i = gt_kernels[:, i, :, :]
|
||||
loss_kernel_i = self.dice_loss(kernel_i, gt_kernel_i, selected_masks)
|
||||
loss_kernels.append(loss_kernel_i)
|
||||
loss_kernels = paddle.mean(paddle.stack(loss_kernels, axis=1), axis=1)
|
||||
iou_kernel = iou(
|
||||
(kernels[:, -1, :, :] > 0).astype("int64"),
|
||||
gt_kernels[:, -1, :, :],
|
||||
training_masks * gt_texts,
|
||||
reduce=False,
|
||||
)
|
||||
losses.update(dict(loss_kernels=loss_kernels, iou_kernel=iou_kernel))
|
||||
loss = self.alpha * loss_text + (1 - self.alpha) * loss_kernels
|
||||
losses["loss"] = loss
|
||||
if self.reduction == "sum":
|
||||
losses = {x: paddle.sum(v) for x, v in losses.items()}
|
||||
elif self.reduction == "mean":
|
||||
losses = {x: paddle.mean(v) for x, v in losses.items()}
|
||||
return losses
|
||||
|
||||
def dice_loss(self, input, target, mask):
|
||||
input = F.sigmoid(input)
|
||||
|
||||
input = input.reshape([input.shape[0], -1])
|
||||
target = target.reshape([target.shape[0], -1])
|
||||
mask = mask.reshape([mask.shape[0], -1])
|
||||
|
||||
input = input * mask
|
||||
target = target * mask
|
||||
|
||||
a = paddle.sum(input * target, 1)
|
||||
b = paddle.sum(input * input, 1) + self.eps
|
||||
c = paddle.sum(target * target, 1) + self.eps
|
||||
d = (2 * a) / (b + c)
|
||||
return 1 - d
|
||||
|
||||
def ohem_single(self, score, gt_text, training_mask, ohem_ratio=3):
|
||||
pos_num = int(paddle.sum((gt_text > 0.5).astype("float32"))) - int(
|
||||
paddle.sum(
|
||||
paddle.logical_and((gt_text > 0.5), (training_mask <= 0.5)).astype(
|
||||
"float32"
|
||||
)
|
||||
)
|
||||
)
|
||||
|
||||
if pos_num == 0:
|
||||
selected_mask = training_mask
|
||||
selected_mask = selected_mask.reshape(
|
||||
[1, selected_mask.shape[0], selected_mask.shape[1]]
|
||||
).astype("float32")
|
||||
return selected_mask
|
||||
|
||||
neg_num = int(paddle.sum((gt_text <= 0.5).astype("float32")))
|
||||
neg_num = int(min(pos_num * ohem_ratio, neg_num))
|
||||
|
||||
if neg_num == 0:
|
||||
selected_mask = training_mask
|
||||
selected_mask = selected_mask.reshape(
|
||||
[1, selected_mask.shape[0], selected_mask.shape[1]]
|
||||
).astype("float32")
|
||||
return selected_mask
|
||||
|
||||
neg_score = paddle.masked_select(score, gt_text <= 0.5)
|
||||
neg_score_sorted = paddle.sort(-neg_score)
|
||||
threshold = -neg_score_sorted[neg_num - 1]
|
||||
|
||||
selected_mask = paddle.logical_and(
|
||||
paddle.logical_or((score >= threshold), (gt_text > 0.5)),
|
||||
(training_mask > 0.5),
|
||||
)
|
||||
selected_mask = selected_mask.reshape(
|
||||
[1, selected_mask.shape[0], selected_mask.shape[1]]
|
||||
).astype("float32")
|
||||
return selected_mask
|
||||
|
||||
def ohem_batch(self, scores, gt_texts, training_masks, ohem_ratio=3):
|
||||
selected_masks = []
|
||||
for i in range(scores.shape[0]):
|
||||
selected_masks.append(
|
||||
self.ohem_single(
|
||||
scores[i, :, :],
|
||||
gt_texts[i, :, :],
|
||||
training_masks[i, :, :],
|
||||
ohem_ratio,
|
||||
)
|
||||
)
|
||||
|
||||
selected_masks = paddle.concat(selected_masks, 0).astype("float32")
|
||||
return selected_masks
|
||||
133
ppocr/losses/det_sast_loss.py
Normal file
133
ppocr/losses/det_sast_loss.py
Normal file
@@ -0,0 +1,133 @@
|
||||
# copyright (c) 2019 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
from .det_basic_loss import DiceLoss
|
||||
import numpy as np
|
||||
|
||||
|
||||
class SASTLoss(nn.Layer):
|
||||
""" """
|
||||
|
||||
def __init__(self, eps=1e-6, **kwargs):
|
||||
super(SASTLoss, self).__init__()
|
||||
self.dice_loss = DiceLoss(eps=eps)
|
||||
|
||||
def forward(self, predicts, labels):
|
||||
"""
|
||||
tcl_pos: N x 128 x 3
|
||||
tcl_mask: N x 128 x 1
|
||||
tcl_label: N x X list or LoDTensor
|
||||
"""
|
||||
|
||||
f_score = predicts["f_score"]
|
||||
f_border = predicts["f_border"]
|
||||
f_tvo = predicts["f_tvo"]
|
||||
f_tco = predicts["f_tco"]
|
||||
|
||||
l_score, l_border, l_mask, l_tvo, l_tco = labels[1:]
|
||||
|
||||
# score_loss
|
||||
intersection = paddle.sum(f_score * l_score * l_mask)
|
||||
union = paddle.sum(f_score * l_mask) + paddle.sum(l_score * l_mask)
|
||||
score_loss = 1.0 - 2 * intersection / (union + 1e-5)
|
||||
|
||||
# border loss
|
||||
l_border_split, l_border_norm = paddle.split(
|
||||
l_border, num_or_sections=[4, 1], axis=1
|
||||
)
|
||||
f_border_split = f_border
|
||||
border_ex_shape = l_border_norm.shape * np.array([1, 4, 1, 1])
|
||||
l_border_norm_split = paddle.expand(x=l_border_norm, shape=border_ex_shape)
|
||||
l_border_score = paddle.expand(x=l_score, shape=border_ex_shape)
|
||||
l_border_mask = paddle.expand(x=l_mask, shape=border_ex_shape)
|
||||
|
||||
border_diff = l_border_split - f_border_split
|
||||
abs_border_diff = paddle.abs(border_diff)
|
||||
border_sign = abs_border_diff < 1.0
|
||||
border_sign = paddle.cast(border_sign, dtype="float32")
|
||||
border_sign.stop_gradient = True
|
||||
border_in_loss = 0.5 * abs_border_diff * abs_border_diff * border_sign + (
|
||||
abs_border_diff - 0.5
|
||||
) * (1.0 - border_sign)
|
||||
border_out_loss = l_border_norm_split * border_in_loss
|
||||
border_loss = paddle.sum(border_out_loss * l_border_score * l_border_mask) / (
|
||||
paddle.sum(l_border_score * l_border_mask) + 1e-5
|
||||
)
|
||||
|
||||
# tvo_loss
|
||||
l_tvo_split, l_tvo_norm = paddle.split(l_tvo, num_or_sections=[8, 1], axis=1)
|
||||
f_tvo_split = f_tvo
|
||||
tvo_ex_shape = l_tvo_norm.shape * np.array([1, 8, 1, 1])
|
||||
l_tvo_norm_split = paddle.expand(x=l_tvo_norm, shape=tvo_ex_shape)
|
||||
l_tvo_score = paddle.expand(x=l_score, shape=tvo_ex_shape)
|
||||
l_tvo_mask = paddle.expand(x=l_mask, shape=tvo_ex_shape)
|
||||
#
|
||||
tvo_geo_diff = l_tvo_split - f_tvo_split
|
||||
abs_tvo_geo_diff = paddle.abs(tvo_geo_diff)
|
||||
tvo_sign = abs_tvo_geo_diff < 1.0
|
||||
tvo_sign = paddle.cast(tvo_sign, dtype="float32")
|
||||
tvo_sign.stop_gradient = True
|
||||
tvo_in_loss = 0.5 * abs_tvo_geo_diff * abs_tvo_geo_diff * tvo_sign + (
|
||||
abs_tvo_geo_diff - 0.5
|
||||
) * (1.0 - tvo_sign)
|
||||
tvo_out_loss = l_tvo_norm_split * tvo_in_loss
|
||||
tvo_loss = paddle.sum(tvo_out_loss * l_tvo_score * l_tvo_mask) / (
|
||||
paddle.sum(l_tvo_score * l_tvo_mask) + 1e-5
|
||||
)
|
||||
|
||||
# tco_loss
|
||||
l_tco_split, l_tco_norm = paddle.split(l_tco, num_or_sections=[2, 1], axis=1)
|
||||
f_tco_split = f_tco
|
||||
tco_ex_shape = l_tco_norm.shape * np.array([1, 2, 1, 1])
|
||||
l_tco_norm_split = paddle.expand(x=l_tco_norm, shape=tco_ex_shape)
|
||||
l_tco_score = paddle.expand(x=l_score, shape=tco_ex_shape)
|
||||
l_tco_mask = paddle.expand(x=l_mask, shape=tco_ex_shape)
|
||||
|
||||
tco_geo_diff = l_tco_split - f_tco_split
|
||||
abs_tco_geo_diff = paddle.abs(tco_geo_diff)
|
||||
tco_sign = abs_tco_geo_diff < 1.0
|
||||
tco_sign = paddle.cast(tco_sign, dtype="float32")
|
||||
tco_sign.stop_gradient = True
|
||||
tco_in_loss = 0.5 * abs_tco_geo_diff * abs_tco_geo_diff * tco_sign + (
|
||||
abs_tco_geo_diff - 0.5
|
||||
) * (1.0 - tco_sign)
|
||||
tco_out_loss = l_tco_norm_split * tco_in_loss
|
||||
tco_loss = paddle.sum(tco_out_loss * l_tco_score * l_tco_mask) / (
|
||||
paddle.sum(l_tco_score * l_tco_mask) + 1e-5
|
||||
)
|
||||
|
||||
# total loss
|
||||
tvo_lw, tco_lw = 1.5, 1.5
|
||||
score_lw, border_lw = 1.0, 1.0
|
||||
total_loss = (
|
||||
score_loss * score_lw
|
||||
+ border_loss * border_lw
|
||||
+ tvo_loss * tvo_lw
|
||||
+ tco_loss * tco_lw
|
||||
)
|
||||
|
||||
losses = {
|
||||
"loss": total_loss,
|
||||
"score_loss": score_loss,
|
||||
"border_loss": border_loss,
|
||||
"tvo_loss": tvo_loss,
|
||||
"tco_loss": tco_loss,
|
||||
}
|
||||
return losses
|
||||
1192
ppocr/losses/distillation_loss.py
Normal file
1192
ppocr/losses/distillation_loss.py
Normal file
File diff suppressed because it is too large
Load Diff
165
ppocr/losses/e2e_pg_loss.py
Normal file
165
ppocr/losses/e2e_pg_loss.py
Normal file
@@ -0,0 +1,165 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
from paddle import nn
|
||||
import paddle
|
||||
|
||||
from .det_basic_loss import DiceLoss
|
||||
from ppocr.utils.e2e_utils.extract_batchsize import pre_process
|
||||
|
||||
|
||||
class PGLoss(nn.Layer):
|
||||
def __init__(
|
||||
self, tcl_bs, max_text_length, max_text_nums, pad_num, eps=1e-6, **kwargs
|
||||
):
|
||||
super(PGLoss, self).__init__()
|
||||
self.tcl_bs = tcl_bs
|
||||
self.max_text_nums = max_text_nums
|
||||
self.max_text_length = max_text_length
|
||||
self.pad_num = pad_num
|
||||
self.dice_loss = DiceLoss(eps=eps)
|
||||
|
||||
def border_loss(self, f_border, l_border, l_score, l_mask):
|
||||
l_border_split, l_border_norm = paddle.tensor.split(
|
||||
l_border, num_or_sections=[4, 1], axis=1
|
||||
)
|
||||
f_border_split = f_border
|
||||
b, c, h, w = l_border_norm.shape
|
||||
l_border_norm_split = paddle.expand(x=l_border_norm, shape=[b, 4 * c, h, w])
|
||||
b, c, h, w = l_score.shape
|
||||
l_border_score = paddle.expand(x=l_score, shape=[b, 4 * c, h, w])
|
||||
b, c, h, w = l_mask.shape
|
||||
l_border_mask = paddle.expand(x=l_mask, shape=[b, 4 * c, h, w])
|
||||
border_diff = l_border_split - f_border_split
|
||||
abs_border_diff = paddle.abs(border_diff)
|
||||
border_sign = abs_border_diff < 1.0
|
||||
border_sign = paddle.cast(border_sign, dtype="float32")
|
||||
border_sign.stop_gradient = True
|
||||
border_in_loss = 0.5 * abs_border_diff * abs_border_diff * border_sign + (
|
||||
abs_border_diff - 0.5
|
||||
) * (1.0 - border_sign)
|
||||
border_out_loss = l_border_norm_split * border_in_loss
|
||||
border_loss = paddle.sum(border_out_loss * l_border_score * l_border_mask) / (
|
||||
paddle.sum(l_border_score * l_border_mask) + 1e-5
|
||||
)
|
||||
return border_loss
|
||||
|
||||
def direction_loss(self, f_direction, l_direction, l_score, l_mask):
|
||||
l_direction_split, l_direction_norm = paddle.tensor.split(
|
||||
l_direction, num_or_sections=[2, 1], axis=1
|
||||
)
|
||||
f_direction_split = f_direction
|
||||
b, c, h, w = l_direction_norm.shape
|
||||
l_direction_norm_split = paddle.expand(
|
||||
x=l_direction_norm, shape=[b, 2 * c, h, w]
|
||||
)
|
||||
b, c, h, w = l_score.shape
|
||||
l_direction_score = paddle.expand(x=l_score, shape=[b, 2 * c, h, w])
|
||||
b, c, h, w = l_mask.shape
|
||||
l_direction_mask = paddle.expand(x=l_mask, shape=[b, 2 * c, h, w])
|
||||
direction_diff = l_direction_split - f_direction_split
|
||||
abs_direction_diff = paddle.abs(direction_diff)
|
||||
direction_sign = abs_direction_diff < 1.0
|
||||
direction_sign = paddle.cast(direction_sign, dtype="float32")
|
||||
direction_sign.stop_gradient = True
|
||||
direction_in_loss = (
|
||||
0.5 * abs_direction_diff * abs_direction_diff * direction_sign
|
||||
+ (abs_direction_diff - 0.5) * (1.0 - direction_sign)
|
||||
)
|
||||
direction_out_loss = l_direction_norm_split * direction_in_loss
|
||||
direction_loss = paddle.sum(
|
||||
direction_out_loss * l_direction_score * l_direction_mask
|
||||
) / (paddle.sum(l_direction_score * l_direction_mask) + 1e-5)
|
||||
return direction_loss
|
||||
|
||||
def ctcloss(self, f_char, tcl_pos, tcl_mask, tcl_label, label_t):
|
||||
f_char = paddle.transpose(f_char, [0, 2, 3, 1])
|
||||
tcl_pos = paddle.reshape(tcl_pos, [-1, 3])
|
||||
tcl_pos = paddle.cast(tcl_pos, dtype=int)
|
||||
f_tcl_char = paddle.gather_nd(f_char, tcl_pos)
|
||||
f_tcl_char = paddle.reshape(
|
||||
f_tcl_char, [-1, 64, self.pad_num + 1]
|
||||
) # len(Lexicon_Table)+1
|
||||
f_tcl_char_fg, f_tcl_char_bg = paddle.split(
|
||||
f_tcl_char, [self.pad_num, 1], axis=2
|
||||
)
|
||||
f_tcl_char_bg = f_tcl_char_bg * tcl_mask + (1.0 - tcl_mask) * 20.0
|
||||
b, c, l = tcl_mask.shape
|
||||
tcl_mask_fg = paddle.expand(x=tcl_mask, shape=[b, c, self.pad_num * l])
|
||||
tcl_mask_fg.stop_gradient = True
|
||||
f_tcl_char_fg = f_tcl_char_fg * tcl_mask_fg + (1.0 - tcl_mask_fg) * (-20.0)
|
||||
f_tcl_char_mask = paddle.concat([f_tcl_char_fg, f_tcl_char_bg], axis=2)
|
||||
f_tcl_char_ld = paddle.transpose(f_tcl_char_mask, (1, 0, 2))
|
||||
N, B, _ = f_tcl_char_ld.shape
|
||||
input_lengths = paddle.to_tensor([N] * B, dtype="int64")
|
||||
cost = paddle.nn.functional.ctc_loss(
|
||||
log_probs=f_tcl_char_ld,
|
||||
labels=tcl_label,
|
||||
input_lengths=input_lengths,
|
||||
label_lengths=label_t,
|
||||
blank=self.pad_num,
|
||||
reduction="none",
|
||||
)
|
||||
cost = cost.mean()
|
||||
return cost
|
||||
|
||||
def forward(self, predicts, labels):
|
||||
(
|
||||
images,
|
||||
tcl_maps,
|
||||
tcl_label_maps,
|
||||
border_maps,
|
||||
direction_maps,
|
||||
training_masks,
|
||||
label_list,
|
||||
pos_list,
|
||||
pos_mask,
|
||||
) = labels
|
||||
# for all the batch_size
|
||||
pos_list, pos_mask, label_list, label_t = pre_process(
|
||||
label_list,
|
||||
pos_list,
|
||||
pos_mask,
|
||||
self.max_text_length,
|
||||
self.max_text_nums,
|
||||
self.pad_num,
|
||||
self.tcl_bs,
|
||||
)
|
||||
|
||||
f_score, f_border, f_direction, f_char = (
|
||||
predicts["f_score"],
|
||||
predicts["f_border"],
|
||||
predicts["f_direction"],
|
||||
predicts["f_char"],
|
||||
)
|
||||
score_loss = self.dice_loss(f_score, tcl_maps, training_masks)
|
||||
border_loss = self.border_loss(f_border, border_maps, tcl_maps, training_masks)
|
||||
direction_loss = self.direction_loss(
|
||||
f_direction, direction_maps, tcl_maps, training_masks
|
||||
)
|
||||
ctc_loss = self.ctcloss(f_char, pos_list, pos_mask, label_list, label_t)
|
||||
loss_all = score_loss + border_loss + direction_loss + 5 * ctc_loss
|
||||
|
||||
losses = {
|
||||
"loss": loss_all,
|
||||
"score_loss": score_loss,
|
||||
"border_loss": border_loss,
|
||||
"direction_loss": direction_loss,
|
||||
"ctc_loss": ctc_loss,
|
||||
}
|
||||
return losses
|
||||
116
ppocr/losses/kie_sdmgr_loss.py
Normal file
116
ppocr/losses/kie_sdmgr_loss.py
Normal file
@@ -0,0 +1,116 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
# reference from : https://github.com/open-mmlab/mmocr/blob/main/mmocr/models/kie/losses/sdmgr_loss.py
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
from paddle import nn
|
||||
import paddle
|
||||
|
||||
|
||||
class SDMGRLoss(nn.Layer):
|
||||
def __init__(self, node_weight=1.0, edge_weight=1.0, ignore=0):
|
||||
super().__init__()
|
||||
self.loss_node = nn.CrossEntropyLoss(ignore_index=ignore)
|
||||
self.loss_edge = nn.CrossEntropyLoss(ignore_index=-1)
|
||||
self.node_weight = node_weight
|
||||
self.edge_weight = edge_weight
|
||||
self.ignore = ignore
|
||||
|
||||
def pre_process(self, gts, tag):
|
||||
gts, tag = gts.numpy(), tag.numpy().tolist()
|
||||
temp_gts = []
|
||||
batch = len(tag)
|
||||
for i in range(batch):
|
||||
num, recoder_len = tag[i][0], tag[i][1]
|
||||
temp_gts.append(paddle.to_tensor(gts[i, :num, : num + 1], dtype="int64"))
|
||||
return temp_gts
|
||||
|
||||
def accuracy(self, pred, target, topk=1, thresh=None):
|
||||
"""Calculate accuracy according to the prediction and target.
|
||||
|
||||
Args:
|
||||
pred (torch.Tensor): The model prediction, shape (N, num_class)
|
||||
target (torch.Tensor): The target of each prediction, shape (N, )
|
||||
topk (int | tuple[int], optional): If the predictions in ``topk``
|
||||
matches the target, the predictions will be regarded as
|
||||
correct ones. Defaults to 1.
|
||||
thresh (float, optional): If not None, predictions with scores under
|
||||
this threshold are considered incorrect. Default to None.
|
||||
|
||||
Returns:
|
||||
float | tuple[float]: If the input ``topk`` is a single integer,
|
||||
the function will return a single float as accuracy. If
|
||||
``topk`` is a tuple containing multiple integers, the
|
||||
function will return a tuple containing accuracies of
|
||||
each ``topk`` number.
|
||||
"""
|
||||
assert isinstance(topk, (int, tuple))
|
||||
if isinstance(topk, int):
|
||||
topk = (topk,)
|
||||
return_single = True
|
||||
else:
|
||||
return_single = False
|
||||
|
||||
maxk = max(topk)
|
||||
if pred.shape[0] == 0:
|
||||
accu = [pred.new_tensor(0.0) for i in range(len(topk))]
|
||||
return accu[0] if return_single else accu
|
||||
pred_value, pred_label = paddle.topk(pred, maxk, axis=1)
|
||||
pred_label = pred_label.transpose([1, 0]) # transpose to shape (maxk, N)
|
||||
correct = paddle.equal(
|
||||
pred_label, (target.reshape([1, -1]).expand_as(pred_label))
|
||||
)
|
||||
res = []
|
||||
for k in topk:
|
||||
correct_k = paddle.sum(
|
||||
correct[:k].reshape([-1]).astype("float32"), axis=0, keepdim=True
|
||||
)
|
||||
res.append(
|
||||
paddle.multiply(correct_k, paddle.to_tensor(100.0 / pred.shape[0]))
|
||||
)
|
||||
return res[0] if return_single else res
|
||||
|
||||
def forward(self, pred, batch):
|
||||
node_preds, edge_preds = pred
|
||||
gts, tag = batch[4], batch[5]
|
||||
gts = self.pre_process(gts, tag)
|
||||
node_gts, edge_gts = [], []
|
||||
for gt in gts:
|
||||
node_gts.append(gt[:, 0])
|
||||
edge_gts.append(gt[:, 1:].reshape([-1]))
|
||||
node_gts = paddle.concat(node_gts)
|
||||
edge_gts = paddle.concat(edge_gts)
|
||||
|
||||
node_valids = paddle.nonzero(node_gts != self.ignore).reshape([-1])
|
||||
edge_valids = paddle.nonzero(edge_gts != -1).reshape([-1])
|
||||
loss_node = self.loss_node(node_preds, node_gts)
|
||||
loss_edge = self.loss_edge(edge_preds, edge_gts)
|
||||
loss = self.node_weight * loss_node + self.edge_weight * loss_edge
|
||||
return dict(
|
||||
loss=loss,
|
||||
loss_node=loss_node,
|
||||
loss_edge=loss_edge,
|
||||
acc_node=self.accuracy(
|
||||
paddle.gather(node_preds, node_valids),
|
||||
paddle.gather(node_gts, node_valids),
|
||||
),
|
||||
acc_edge=self.accuracy(
|
||||
paddle.gather(edge_preds, edge_valids),
|
||||
paddle.gather(edge_gts, edge_valids),
|
||||
),
|
||||
)
|
||||
103
ppocr/losses/rec_aster_loss.py
Normal file
103
ppocr/losses/rec_aster_loss.py
Normal file
@@ -0,0 +1,103 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
|
||||
|
||||
class CosineEmbeddingLoss(nn.Layer):
|
||||
def __init__(self, margin=0.0):
|
||||
super(CosineEmbeddingLoss, self).__init__()
|
||||
self.margin = margin
|
||||
self.epsilon = 1e-12
|
||||
|
||||
def forward(self, x1, x2, target):
|
||||
similarity = paddle.sum(x1 * x2, axis=-1) / (
|
||||
paddle.norm(x1, axis=-1) * paddle.norm(x2, axis=-1) + self.epsilon
|
||||
)
|
||||
one_list = paddle.full_like(target, fill_value=1)
|
||||
out = paddle.mean(
|
||||
paddle.where(
|
||||
paddle.equal(target, one_list),
|
||||
1.0 - similarity,
|
||||
paddle.maximum(paddle.zeros_like(similarity), similarity - self.margin),
|
||||
)
|
||||
)
|
||||
|
||||
return out
|
||||
|
||||
|
||||
class AsterLoss(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
weight=None,
|
||||
size_average=True,
|
||||
ignore_index=-100,
|
||||
sequence_normalize=False,
|
||||
sample_normalize=True,
|
||||
**kwargs,
|
||||
):
|
||||
super(AsterLoss, self).__init__()
|
||||
self.weight = weight
|
||||
self.size_average = size_average
|
||||
self.ignore_index = ignore_index
|
||||
self.sequence_normalize = sequence_normalize
|
||||
self.sample_normalize = sample_normalize
|
||||
self.loss_sem = CosineEmbeddingLoss()
|
||||
self.is_cosin_loss = True
|
||||
self.loss_func_rec = nn.CrossEntropyLoss(weight=None, reduction="none")
|
||||
|
||||
def forward(self, predicts, batch):
|
||||
targets = batch[1].astype("int64")
|
||||
label_lengths = batch[2].astype("int64")
|
||||
sem_target = batch[3].astype("float32")
|
||||
embedding_vectors = predicts["embedding_vectors"]
|
||||
rec_pred = predicts["rec_pred"]
|
||||
|
||||
if not self.is_cosin_loss:
|
||||
sem_loss = paddle.sum(self.loss_sem(embedding_vectors, sem_target))
|
||||
else:
|
||||
label_target = paddle.ones([embedding_vectors.shape[0]])
|
||||
sem_loss = paddle.sum(
|
||||
self.loss_sem(embedding_vectors, sem_target, label_target)
|
||||
)
|
||||
|
||||
# rec loss
|
||||
batch_size, def_max_length = targets.shape[0], targets.shape[1]
|
||||
|
||||
mask = paddle.zeros([batch_size, def_max_length])
|
||||
for i in range(batch_size):
|
||||
mask[i, : label_lengths[i]] = 1
|
||||
mask = paddle.cast(mask, "float32")
|
||||
max_length = max(label_lengths)
|
||||
assert max_length == rec_pred.shape[1]
|
||||
targets = targets[:, :max_length]
|
||||
mask = mask[:, :max_length]
|
||||
rec_pred = paddle.reshape(rec_pred, [-1, rec_pred.shape[2]])
|
||||
input = nn.functional.log_softmax(rec_pred, axis=1)
|
||||
targets = paddle.reshape(targets, [-1, 1])
|
||||
mask = paddle.reshape(mask, [-1, 1])
|
||||
output = -paddle.index_sample(input, index=targets) * mask
|
||||
output = paddle.sum(output)
|
||||
if self.sequence_normalize:
|
||||
output = output / paddle.sum(mask)
|
||||
if self.sample_normalize:
|
||||
output = output / batch_size
|
||||
|
||||
loss = output + sem_loss * 0.1
|
||||
return {"loss": loss}
|
||||
43
ppocr/losses/rec_att_loss.py
Normal file
43
ppocr/losses/rec_att_loss.py
Normal file
@@ -0,0 +1,43 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
|
||||
|
||||
class AttentionLoss(nn.Layer):
|
||||
def __init__(self, **kwargs):
|
||||
super(AttentionLoss, self).__init__()
|
||||
self.loss_func = nn.CrossEntropyLoss(weight=None, reduction="none")
|
||||
|
||||
def forward(self, predicts, batch):
|
||||
targets = batch[1].astype("int64")
|
||||
label_lengths = batch[2].astype("int64")
|
||||
batch_size, num_steps, num_classes = (
|
||||
predicts.shape[0],
|
||||
predicts.shape[1],
|
||||
predicts.shape[2],
|
||||
)
|
||||
assert (
|
||||
len(targets.shape) == len(list(predicts.shape)) - 1
|
||||
), "The target's shape and inputs's shape is [N, d] and [N, num_steps]"
|
||||
|
||||
inputs = paddle.reshape(predicts, [-1, predicts.shape[-1]])
|
||||
targets = paddle.reshape(targets, [-1])
|
||||
|
||||
return {"loss": paddle.sum(self.loss_func(inputs, targets))}
|
||||
88
ppocr/losses/rec_can_loss.py
Normal file
88
ppocr/losses/rec_can_loss.py
Normal file
@@ -0,0 +1,88 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/LBH1024/CAN/models/can.py
|
||||
"""
|
||||
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
import numpy as np
|
||||
|
||||
|
||||
class CANLoss(nn.Layer):
|
||||
"""
|
||||
CANLoss is consist of two part:
|
||||
word_average_loss: average accuracy of the symbol
|
||||
counting_loss: counting loss of every symbol
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
super(CANLoss, self).__init__()
|
||||
|
||||
self.use_label_mask = False
|
||||
self.out_channel = 111
|
||||
self.cross = (
|
||||
nn.CrossEntropyLoss(reduction="none")
|
||||
if self.use_label_mask
|
||||
else nn.CrossEntropyLoss()
|
||||
)
|
||||
self.counting_loss = nn.SmoothL1Loss(reduction="mean")
|
||||
self.ratio = 16
|
||||
|
||||
def forward(self, preds, batch):
|
||||
word_probs = preds[0]
|
||||
counting_preds = preds[1]
|
||||
counting_preds1 = preds[2]
|
||||
counting_preds2 = preds[3]
|
||||
labels = batch[2]
|
||||
labels_mask = batch[3]
|
||||
counting_labels = gen_counting_label(labels, self.out_channel, True)
|
||||
counting_loss = (
|
||||
self.counting_loss(counting_preds1, counting_labels)
|
||||
+ self.counting_loss(counting_preds2, counting_labels)
|
||||
+ self.counting_loss(counting_preds, counting_labels)
|
||||
)
|
||||
|
||||
word_loss = self.cross(
|
||||
paddle.reshape(word_probs, [-1, word_probs.shape[-1]]),
|
||||
paddle.reshape(labels, [-1]),
|
||||
)
|
||||
word_average_loss = (
|
||||
paddle.sum(paddle.reshape(word_loss * labels_mask, [-1]))
|
||||
/ (paddle.sum(labels_mask) + 1e-10)
|
||||
if self.use_label_mask
|
||||
else word_loss
|
||||
)
|
||||
loss = word_average_loss + counting_loss
|
||||
return {"loss": loss}
|
||||
|
||||
|
||||
def gen_counting_label(labels, channel, tag):
|
||||
b, t = labels.shape
|
||||
counting_labels = np.zeros([b, channel])
|
||||
|
||||
if tag:
|
||||
ignore = [0, 1, 107, 108, 109, 110]
|
||||
else:
|
||||
ignore = []
|
||||
for i in range(b):
|
||||
for j in range(t):
|
||||
k = labels[i][j]
|
||||
if k in ignore:
|
||||
continue
|
||||
else:
|
||||
counting_labels[i][k] += 1
|
||||
counting_labels = paddle.to_tensor(counting_labels, dtype="float32")
|
||||
return counting_labels
|
||||
61
ppocr/losses/rec_ce_loss.py
Normal file
61
ppocr/losses/rec_ce_loss.py
Normal file
@@ -0,0 +1,61 @@
|
||||
import paddle
|
||||
from paddle import nn
|
||||
import paddle.nn.functional as F
|
||||
|
||||
|
||||
class CELoss(nn.Layer):
|
||||
def __init__(self, smoothing=False, with_all=False, ignore_index=-1, **kwargs):
|
||||
super(CELoss, self).__init__()
|
||||
if ignore_index >= 0:
|
||||
self.loss_func = nn.CrossEntropyLoss(
|
||||
reduction="mean", ignore_index=ignore_index
|
||||
)
|
||||
else:
|
||||
self.loss_func = nn.CrossEntropyLoss(reduction="mean")
|
||||
self.smoothing = smoothing
|
||||
self.with_all = with_all
|
||||
|
||||
def forward(self, pred, batch):
|
||||
if isinstance(pred, dict): # for ABINet
|
||||
loss = {}
|
||||
loss_sum = []
|
||||
for name, logits in pred.items():
|
||||
if isinstance(logits, list):
|
||||
logit_num = len(logits)
|
||||
all_tgt = paddle.concat([batch[1]] * logit_num, 0)
|
||||
all_logits = paddle.concat(logits, 0)
|
||||
flt_logtis = all_logits.reshape([-1, all_logits.shape[2]])
|
||||
flt_tgt = all_tgt.reshape([-1])
|
||||
else:
|
||||
flt_logtis = logits.reshape([-1, logits.shape[2]])
|
||||
flt_tgt = batch[1].reshape([-1])
|
||||
loss[name + "_loss"] = self.loss_func(flt_logtis, flt_tgt)
|
||||
loss_sum.append(loss[name + "_loss"])
|
||||
loss["loss"] = sum(loss_sum)
|
||||
return loss
|
||||
else:
|
||||
if self.with_all: # for ViTSTR
|
||||
tgt = batch[1]
|
||||
pred = pred.reshape([-1, pred.shape[2]])
|
||||
tgt = tgt.reshape([-1])
|
||||
loss = self.loss_func(pred, tgt)
|
||||
return {"loss": loss}
|
||||
else: # for NRTR
|
||||
max_len = batch[2].max()
|
||||
tgt = batch[1][:, 1 : 2 + max_len]
|
||||
pred = pred.reshape([-1, pred.shape[2]])
|
||||
tgt = tgt.reshape([-1])
|
||||
if self.smoothing:
|
||||
eps = 0.1
|
||||
n_class = pred.shape[1]
|
||||
one_hot = F.one_hot(tgt, pred.shape[1])
|
||||
one_hot = one_hot * (1 - eps) + (1 - one_hot) * eps / (n_class - 1)
|
||||
log_prb = F.log_softmax(pred, axis=1)
|
||||
non_pad_mask = paddle.not_equal(
|
||||
tgt, paddle.zeros(tgt.shape, dtype=tgt.dtype)
|
||||
)
|
||||
loss = -(one_hot * log_prb).sum(axis=1)
|
||||
loss = loss.masked_select(non_pad_mask).mean()
|
||||
else:
|
||||
loss = self.loss_func(pred, tgt)
|
||||
return {"loss": loss}
|
||||
76
ppocr/losses/rec_cppd_loss.py
Executable file
76
ppocr/losses/rec_cppd_loss.py
Executable file
@@ -0,0 +1,76 @@
|
||||
# copyright (c) 2023 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
import paddle.nn.functional as F
|
||||
|
||||
|
||||
class CPPDLoss(nn.Layer):
|
||||
def __init__(
|
||||
self, smoothing=False, ignore_index=100, sideloss_weight=1.0, **kwargs
|
||||
):
|
||||
super(CPPDLoss, self).__init__()
|
||||
self.edge_ce = nn.CrossEntropyLoss(reduction="mean", ignore_index=ignore_index)
|
||||
self.char_node_ce = nn.CrossEntropyLoss(reduction="mean")
|
||||
self.pos_node_ce = nn.BCEWithLogitsLoss(reduction="mean")
|
||||
self.smoothing = smoothing
|
||||
self.ignore_index = ignore_index
|
||||
self.sideloss_weight = sideloss_weight
|
||||
|
||||
def label_smoothing_ce(self, preds, targets):
|
||||
non_pad_mask = paddle.not_equal(
|
||||
targets,
|
||||
paddle.zeros(targets.shape, dtype=targets.dtype) + self.ignore_index,
|
||||
)
|
||||
tgts = paddle.where(
|
||||
targets
|
||||
== (paddle.zeros(targets.shape, dtype=targets.dtype) + self.ignore_index),
|
||||
paddle.zeros(targets.shape, dtype=targets.dtype),
|
||||
targets,
|
||||
)
|
||||
eps = 0.1
|
||||
n_class = preds.shape[1]
|
||||
one_hot = F.one_hot(tgts, preds.shape[1])
|
||||
one_hot = one_hot * (1 - eps) + (1 - one_hot) * eps / (n_class - 1)
|
||||
log_prb = F.log_softmax(preds, axis=1)
|
||||
loss = -(one_hot * log_prb).sum(axis=1)
|
||||
loss = loss.masked_select(non_pad_mask).mean()
|
||||
return loss
|
||||
|
||||
def forward(self, pred, batch):
|
||||
node_feats, edge_feats = pred
|
||||
node_tgt = batch[2]
|
||||
char_tgt = batch[1]
|
||||
|
||||
loss_char_node = self.char_node_ce(
|
||||
node_feats[0].flatten(0, 1), node_tgt[:, :-26].flatten(0, 1)
|
||||
)
|
||||
loss_pos_node = self.pos_node_ce(
|
||||
node_feats[1].flatten(0, 1), node_tgt[:, -26:].flatten(0, 1).cast("float32")
|
||||
)
|
||||
loss_node = loss_char_node + loss_pos_node
|
||||
|
||||
edge_feats = edge_feats.flatten(0, 1)
|
||||
char_tgt = char_tgt.flatten(0, 1)
|
||||
if self.smoothing:
|
||||
loss_edge = self.label_smoothing_ce(edge_feats, char_tgt)
|
||||
else:
|
||||
loss_edge = self.edge_ce(edge_feats, char_tgt)
|
||||
|
||||
return {
|
||||
"loss": self.sideloss_weight * loss_node + loss_edge,
|
||||
"loss_node": self.sideloss_weight * loss_node,
|
||||
"loss_edge": loss_edge,
|
||||
}
|
||||
46
ppocr/losses/rec_ctc_loss.py
Executable file
46
ppocr/losses/rec_ctc_loss.py
Executable file
@@ -0,0 +1,46 @@
|
||||
# copyright (c) 2019 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
|
||||
|
||||
class CTCLoss(nn.Layer):
|
||||
def __init__(self, use_focal_loss=False, **kwargs):
|
||||
super(CTCLoss, self).__init__()
|
||||
self.loss_func = nn.CTCLoss(blank=0, reduction="none")
|
||||
self.use_focal_loss = use_focal_loss
|
||||
|
||||
def forward(self, predicts, batch):
|
||||
if isinstance(predicts, (list, tuple)):
|
||||
predicts = predicts[-1]
|
||||
predicts = predicts.transpose((1, 0, 2))
|
||||
N, B, _ = predicts.shape
|
||||
preds_lengths = paddle.to_tensor(
|
||||
[N] * B, dtype="int64", place=paddle.CPUPlace()
|
||||
)
|
||||
labels = batch[1].astype("int32")
|
||||
label_lengths = batch[2].astype("int64")
|
||||
loss = self.loss_func(predicts, labels, preds_lengths, label_lengths)
|
||||
if self.use_focal_loss:
|
||||
weight = paddle.exp(-loss)
|
||||
weight = paddle.subtract(paddle.to_tensor([1.0]), weight)
|
||||
weight = paddle.square(weight)
|
||||
loss = paddle.multiply(loss, weight)
|
||||
loss = loss.mean()
|
||||
return {"loss": loss}
|
||||
76
ppocr/losses/rec_enhanced_ctc_loss.py
Normal file
76
ppocr/losses/rec_enhanced_ctc_loss.py
Normal file
@@ -0,0 +1,76 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
from .ace_loss import ACELoss
|
||||
from .center_loss import CenterLoss
|
||||
from .rec_ctc_loss import CTCLoss
|
||||
|
||||
|
||||
class EnhancedCTCLoss(nn.Layer):
|
||||
def __init__(
|
||||
self,
|
||||
use_focal_loss=False,
|
||||
use_ace_loss=False,
|
||||
ace_loss_weight=0.1,
|
||||
use_center_loss=False,
|
||||
center_loss_weight=0.05,
|
||||
num_classes=6625,
|
||||
feat_dim=96,
|
||||
init_center=False,
|
||||
center_file_path=None,
|
||||
**kwargs,
|
||||
):
|
||||
super(EnhancedCTCLoss, self).__init__()
|
||||
self.ctc_loss_func = CTCLoss(use_focal_loss=use_focal_loss)
|
||||
|
||||
self.use_ace_loss = False
|
||||
if use_ace_loss:
|
||||
self.use_ace_loss = use_ace_loss
|
||||
self.ace_loss_func = ACELoss()
|
||||
self.ace_loss_weight = ace_loss_weight
|
||||
|
||||
self.use_center_loss = False
|
||||
if use_center_loss:
|
||||
self.use_center_loss = use_center_loss
|
||||
self.center_loss_func = CenterLoss(
|
||||
num_classes=num_classes,
|
||||
feat_dim=feat_dim,
|
||||
init_center=init_center,
|
||||
center_file_path=center_file_path,
|
||||
)
|
||||
self.center_loss_weight = center_loss_weight
|
||||
|
||||
def __call__(self, predicts, batch):
|
||||
loss = self.ctc_loss_func(predicts, batch)["loss"]
|
||||
|
||||
if self.use_center_loss:
|
||||
center_loss = (
|
||||
self.center_loss_func(predicts, batch)["loss_center"]
|
||||
* self.center_loss_weight
|
||||
)
|
||||
loss = loss + center_loss
|
||||
|
||||
if self.use_ace_loss:
|
||||
ace_loss = (
|
||||
self.ace_loss_func(predicts, batch)["loss_ace"] * self.ace_loss_weight
|
||||
)
|
||||
loss = loss + ace_loss
|
||||
|
||||
return {"enhanced_ctc_loss": loss}
|
||||
47
ppocr/losses/rec_latexocr_loss.py
Normal file
47
ppocr/losses/rec_latexocr_loss.py
Normal file
@@ -0,0 +1,47 @@
|
||||
# copyright (c) 2024 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/lucidrains/x-transformers/blob/main/x_transformers/autoregressive_wrapper.py
|
||||
"""
|
||||
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
import numpy as np
|
||||
|
||||
|
||||
class LaTeXOCRLoss(nn.Layer):
|
||||
"""
|
||||
LaTeXOCR adopt CrossEntropyLoss for network training.
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
super(LaTeXOCRLoss, self).__init__()
|
||||
self.ignore_index = -100
|
||||
self.cross = nn.CrossEntropyLoss(
|
||||
reduction="mean", ignore_index=self.ignore_index
|
||||
)
|
||||
|
||||
def forward(self, preds, batch):
|
||||
word_probs = preds
|
||||
labels = batch[1][:, 1:]
|
||||
word_loss = self.cross(
|
||||
paddle.reshape(word_probs, [-1, word_probs.shape[-1]]),
|
||||
paddle.reshape(labels, [-1]),
|
||||
)
|
||||
|
||||
loss = word_loss
|
||||
return {"loss": loss}
|
||||
68
ppocr/losses/rec_multi_loss.py
Normal file
68
ppocr/losses/rec_multi_loss.py
Normal file
@@ -0,0 +1,68 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
|
||||
from .rec_ctc_loss import CTCLoss
|
||||
from .rec_sar_loss import SARLoss
|
||||
from .rec_nrtr_loss import NRTRLoss
|
||||
|
||||
|
||||
class MultiLoss(nn.Layer):
|
||||
def __init__(self, **kwargs):
|
||||
super().__init__()
|
||||
self.loss_funcs = {}
|
||||
self.loss_list = kwargs.pop("loss_config_list")
|
||||
self.weight_1 = kwargs.get("weight_1", 1.0)
|
||||
self.weight_2 = kwargs.get("weight_2", 1.0)
|
||||
for loss_info in self.loss_list:
|
||||
for name, param in loss_info.items():
|
||||
if param is not None:
|
||||
kwargs.update(param)
|
||||
loss = eval(name)(**kwargs)
|
||||
self.loss_funcs[name] = loss
|
||||
|
||||
def forward(self, predicts, batch):
|
||||
self.total_loss = {}
|
||||
total_loss = 0.0
|
||||
# batch [image, label_ctc, label_sar, length, valid_ratio]
|
||||
for name, loss_func in self.loss_funcs.items():
|
||||
if name == "CTCLoss":
|
||||
loss = (
|
||||
loss_func(predicts["ctc"], batch[:2] + batch[3:])["loss"]
|
||||
* self.weight_1
|
||||
)
|
||||
elif name == "SARLoss":
|
||||
loss = (
|
||||
loss_func(predicts["sar"], batch[:1] + batch[2:])["loss"]
|
||||
* self.weight_2
|
||||
)
|
||||
elif name == "NRTRLoss":
|
||||
loss = (
|
||||
loss_func(predicts["gtc"], batch[:1] + batch[2:])["loss"]
|
||||
* self.weight_2
|
||||
)
|
||||
else:
|
||||
raise NotImplementedError(
|
||||
"{} is not supported in MultiLoss yet".format(name)
|
||||
)
|
||||
self.total_loss[name] = loss
|
||||
total_loss += loss
|
||||
self.total_loss["loss"] = total_loss
|
||||
return self.total_loss
|
||||
33
ppocr/losses/rec_nrtr_loss.py
Normal file
33
ppocr/losses/rec_nrtr_loss.py
Normal file
@@ -0,0 +1,33 @@
|
||||
import paddle
|
||||
from paddle import nn
|
||||
import paddle.nn.functional as F
|
||||
|
||||
|
||||
class NRTRLoss(nn.Layer):
|
||||
def __init__(self, smoothing=True, ignore_index=0, **kwargs):
|
||||
super(NRTRLoss, self).__init__()
|
||||
if ignore_index >= 0 and not smoothing:
|
||||
self.loss_func = nn.CrossEntropyLoss(
|
||||
reduction="mean", ignore_index=ignore_index
|
||||
)
|
||||
self.smoothing = smoothing
|
||||
|
||||
def forward(self, pred, batch):
|
||||
max_len = batch[2].max()
|
||||
tgt = batch[1][:, 1 : 2 + max_len]
|
||||
pred = pred.reshape([-1, pred.shape[2]])
|
||||
tgt = tgt.reshape([-1])
|
||||
if self.smoothing:
|
||||
eps = 0.1
|
||||
n_class = pred.shape[1]
|
||||
one_hot = F.one_hot(tgt, pred.shape[1])
|
||||
one_hot = one_hot * (1 - eps) + (1 - one_hot) * eps / (n_class - 1)
|
||||
log_prb = F.log_softmax(pred, axis=1)
|
||||
non_pad_mask = paddle.not_equal(
|
||||
tgt, paddle.zeros(tgt.shape, dtype=tgt.dtype)
|
||||
)
|
||||
loss = -(one_hot * log_prb).sum(axis=1)
|
||||
loss = loss.masked_select(non_pad_mask).mean()
|
||||
else:
|
||||
loss = self.loss_func(pred, tgt)
|
||||
return {"loss": loss}
|
||||
52
ppocr/losses/rec_parseq_loss.py
Normal file
52
ppocr/losses/rec_parseq_loss.py
Normal file
@@ -0,0 +1,52 @@
|
||||
# copyright (c) 2020 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
|
||||
|
||||
class ParseQLoss(nn.Layer):
|
||||
def __init__(self, **kwargs):
|
||||
super(ParseQLoss, self).__init__()
|
||||
|
||||
def forward(self, predicts, targets):
|
||||
label = targets[1] # label
|
||||
label_len = targets[2]
|
||||
max_step = paddle.max(label_len).cpu().numpy()[0] + 2
|
||||
tgt = label[:, :max_step]
|
||||
|
||||
logits_list = predicts["logits_list"]
|
||||
pad_id = predicts["pad_id"]
|
||||
eos_id = predicts["eos_id"]
|
||||
|
||||
tgt_out = tgt[:, 1:]
|
||||
loss = 0
|
||||
loss_numel = 0
|
||||
n = (tgt_out != pad_id).sum().item()
|
||||
|
||||
for i, logits in enumerate(logits_list):
|
||||
loss += n * paddle.nn.functional.cross_entropy(
|
||||
input=logits, label=tgt_out.flatten(), ignore_index=pad_id
|
||||
)
|
||||
loss_numel += n
|
||||
if i == 1:
|
||||
tgt_out = paddle.where(condition=tgt_out == eos_id, x=pad_id, y=tgt_out)
|
||||
n = (tgt_out != pad_id).sum().item()
|
||||
loss /= loss_numel
|
||||
|
||||
return {"loss": loss}
|
||||
74
ppocr/losses/rec_ppformulanet_loss.py
Normal file
74
ppocr/losses/rec_ppformulanet_loss.py
Normal file
@@ -0,0 +1,74 @@
|
||||
# copyright (c) 2024 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
|
||||
|
||||
class PPFormulaNet_S_Loss(nn.Layer):
|
||||
"""
|
||||
PP=FormulaNet-S adopt CrossEntropyLoss for network training.
|
||||
"""
|
||||
|
||||
def __init__(self, vocab_size=50000, parallel_step=1):
|
||||
super(PPFormulaNet_S_Loss, self).__init__()
|
||||
self.ignore_index = -100
|
||||
self.vocab_size = vocab_size
|
||||
self.parallel_step = int(parallel_step)
|
||||
self.pad_token_id = 1
|
||||
# ignore padding characters during training
|
||||
self.cross = nn.CrossEntropyLoss(
|
||||
reduction="mean", ignore_index=self.ignore_index
|
||||
)
|
||||
|
||||
def forward(self, preds, batch):
|
||||
logits, masked_label = preds
|
||||
|
||||
word_loss = self.cross(
|
||||
paddle.reshape(logits, [-1, logits.shape[-1]]),
|
||||
paddle.reshape(masked_label[:, self.parallel_step :], [-1]),
|
||||
)
|
||||
loss = word_loss
|
||||
return {
|
||||
"loss": loss,
|
||||
"word_loss": word_loss,
|
||||
}
|
||||
|
||||
|
||||
class PPFormulaNet_L_Loss(nn.Layer):
|
||||
"""
|
||||
PPFormulaNet_L adopt CrossEntropyLoss for network training.
|
||||
"""
|
||||
|
||||
def __init__(self, vocab_size=50000):
|
||||
super(PPFormulaNet_L_Loss, self).__init__()
|
||||
self.ignore_index = -100
|
||||
self.vocab_size = vocab_size
|
||||
self.pad_token_id = 1
|
||||
# ignore padding characters during training
|
||||
self.cross = nn.CrossEntropyLoss(
|
||||
reduction="mean", ignore_index=self.ignore_index
|
||||
)
|
||||
|
||||
def forward(self, preds, batch):
|
||||
logits, masked_label = preds
|
||||
|
||||
word_loss = self.cross(
|
||||
paddle.reshape(logits, [-1, logits.shape[-1]]),
|
||||
paddle.reshape(masked_label[:, 1:], [-1]),
|
||||
)
|
||||
loss = word_loss
|
||||
return {
|
||||
"loss": loss,
|
||||
"word_loss": word_loss,
|
||||
}
|
||||
30
ppocr/losses/rec_pren_loss.py
Normal file
30
ppocr/losses/rec_pren_loss.py
Normal file
@@ -0,0 +1,30 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
from paddle import nn
|
||||
|
||||
|
||||
class PRENLoss(nn.Layer):
|
||||
def __init__(self, **kwargs):
|
||||
super(PRENLoss, self).__init__()
|
||||
# note: 0 is padding idx
|
||||
self.loss_func = nn.CrossEntropyLoss(reduction="mean", ignore_index=0)
|
||||
|
||||
def forward(self, predicts, batch):
|
||||
loss = self.loss_func(predicts, batch[1].astype("int64"))
|
||||
return {"loss": loss}
|
||||
70
ppocr/losses/rec_rfl_loss.py
Normal file
70
ppocr/losses/rec_rfl_loss.py
Normal file
@@ -0,0 +1,70 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/hikopensource/DAVAR-Lab-OCR/blob/main/davarocr/davar_common/models/loss/cross_entropy_loss.py
|
||||
"""
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
|
||||
from .basic_loss import CELoss, DistanceLoss
|
||||
|
||||
|
||||
class RFLLoss(nn.Layer):
|
||||
def __init__(self, ignore_index=-100, **kwargs):
|
||||
super().__init__()
|
||||
|
||||
self.cnt_loss = nn.MSELoss(**kwargs)
|
||||
self.seq_loss = nn.CrossEntropyLoss(ignore_index=ignore_index)
|
||||
|
||||
def forward(self, predicts, batch):
|
||||
self.total_loss = {}
|
||||
total_loss = 0.0
|
||||
if isinstance(predicts, tuple) or isinstance(predicts, list):
|
||||
cnt_outputs, seq_outputs = predicts
|
||||
else:
|
||||
cnt_outputs, seq_outputs = predicts, None
|
||||
# batch [image, label, length, cnt_label]
|
||||
if cnt_outputs is not None:
|
||||
cnt_loss = self.cnt_loss(cnt_outputs, paddle.cast(batch[3], paddle.float32))
|
||||
self.total_loss["cnt_loss"] = cnt_loss
|
||||
total_loss += cnt_loss
|
||||
|
||||
if seq_outputs is not None:
|
||||
targets = batch[1].astype("int64")
|
||||
label_lengths = batch[2].astype("int64")
|
||||
batch_size, num_steps, num_classes = (
|
||||
seq_outputs.shape[0],
|
||||
seq_outputs.shape[1],
|
||||
seq_outputs.shape[2],
|
||||
)
|
||||
assert (
|
||||
len(targets.shape) == len(list(seq_outputs.shape)) - 1
|
||||
), "The target's shape and inputs's shape is [N, d] and [N, num_steps]"
|
||||
|
||||
inputs = seq_outputs[:, :-1, :]
|
||||
targets = targets[:, 1:]
|
||||
|
||||
inputs = paddle.reshape(inputs, [-1, inputs.shape[-1]])
|
||||
targets = paddle.reshape(targets, [-1])
|
||||
seq_loss = self.seq_loss(inputs, targets)
|
||||
self.total_loss["seq_loss"] = seq_loss
|
||||
total_loss += seq_loss
|
||||
|
||||
self.total_loss["loss"] = total_loss
|
||||
return self.total_loss
|
||||
36
ppocr/losses/rec_sar_loss.py
Normal file
36
ppocr/losses/rec_sar_loss.py
Normal file
@@ -0,0 +1,36 @@
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
|
||||
|
||||
class SARLoss(nn.Layer):
|
||||
def __init__(self, **kwargs):
|
||||
super(SARLoss, self).__init__()
|
||||
ignore_index = kwargs.get("ignore_index", 92) # 6626
|
||||
self.loss_func = paddle.nn.loss.CrossEntropyLoss(
|
||||
reduction="mean", ignore_index=ignore_index
|
||||
)
|
||||
|
||||
def forward(self, predicts, batch):
|
||||
predict = predicts[
|
||||
:, :-1, :
|
||||
] # ignore last index of outputs to be in same seq_len with targets
|
||||
label = batch[1].astype("int64")[
|
||||
:, 1:
|
||||
] # ignore first index of target in loss calculation
|
||||
batch_size, num_steps, num_classes = (
|
||||
predict.shape[0],
|
||||
predict.shape[1],
|
||||
predict.shape[2],
|
||||
)
|
||||
assert (
|
||||
len(label.shape) == len(list(predict.shape)) - 1
|
||||
), "The target's shape and inputs's shape is [N, d] and [N, num_steps]"
|
||||
|
||||
inputs = paddle.reshape(predict, [-1, num_classes])
|
||||
targets = paddle.reshape(label, [-1])
|
||||
loss = self.loss_func(inputs, targets)
|
||||
return {"loss": loss}
|
||||
53
ppocr/losses/rec_satrn_loss.py
Normal file
53
ppocr/losses/rec_satrn_loss.py
Normal file
@@ -0,0 +1,53 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/open-mmlab/mmocr/blob/1.x/mmocr/models/textrecog/module_losses/ce_module_loss.py
|
||||
"""
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
|
||||
|
||||
class SATRNLoss(nn.Layer):
|
||||
def __init__(self, **kwargs):
|
||||
super(SATRNLoss, self).__init__()
|
||||
ignore_index = kwargs.get("ignore_index", 92) # 6626
|
||||
self.loss_func = paddle.nn.loss.CrossEntropyLoss(
|
||||
reduction="none", ignore_index=ignore_index
|
||||
)
|
||||
|
||||
def forward(self, predicts, batch):
|
||||
predict = predicts[
|
||||
:, :-1, :
|
||||
] # ignore last index of outputs to be in same seq_len with targets
|
||||
label = batch[1].astype("int64")[
|
||||
:, 1:
|
||||
] # ignore first index of target in loss calculation
|
||||
batch_size, num_steps, num_classes = (
|
||||
predict.shape[0],
|
||||
predict.shape[1],
|
||||
predict.shape[2],
|
||||
)
|
||||
assert (
|
||||
len(label.shape) == len(list(predict.shape)) - 1
|
||||
), "The target's shape and inputs's shape is [N, d] and [N, num_steps]"
|
||||
|
||||
inputs = paddle.reshape(predict, [-1, num_classes])
|
||||
targets = paddle.reshape(label, [-1])
|
||||
loss = self.loss_func(inputs, targets)
|
||||
return {"loss": loss.mean()}
|
||||
52
ppocr/losses/rec_spin_att_loss.py
Normal file
52
ppocr/losses/rec_spin_att_loss.py
Normal file
@@ -0,0 +1,52 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
|
||||
"""This code is refer from:
|
||||
https://github.com/hikopensource/DAVAR-Lab-OCR
|
||||
"""
|
||||
|
||||
|
||||
class SPINAttentionLoss(nn.Layer):
|
||||
def __init__(self, reduction="mean", ignore_index=-100, **kwargs):
|
||||
super(SPINAttentionLoss, self).__init__()
|
||||
self.loss_func = nn.CrossEntropyLoss(
|
||||
weight=None, reduction=reduction, ignore_index=ignore_index
|
||||
)
|
||||
|
||||
def forward(self, predicts, batch):
|
||||
targets = batch[1].astype("int64")
|
||||
targets = targets[:, 1:] # remove [eos] in label
|
||||
|
||||
label_lengths = batch[2].astype("int64")
|
||||
batch_size, num_steps, num_classes = (
|
||||
predicts.shape[0],
|
||||
predicts.shape[1],
|
||||
predicts.shape[2],
|
||||
)
|
||||
assert (
|
||||
len(targets.shape) == len(list(predicts.shape)) - 1
|
||||
), "The target's shape and inputs's shape is [N, d] and [N, num_steps]"
|
||||
|
||||
inputs = paddle.reshape(predicts, [-1, predicts.shape[-1]])
|
||||
targets = paddle.reshape(targets, [-1])
|
||||
|
||||
return {"loss": self.loss_func(inputs, targets)}
|
||||
47
ppocr/losses/rec_srn_loss.py
Normal file
47
ppocr/losses/rec_srn_loss.py
Normal file
@@ -0,0 +1,47 @@
|
||||
# copyright (c) 2020 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
|
||||
|
||||
class SRNLoss(nn.Layer):
|
||||
def __init__(self, **kwargs):
|
||||
super(SRNLoss, self).__init__()
|
||||
self.loss_func = paddle.nn.loss.CrossEntropyLoss(reduction="sum")
|
||||
|
||||
def forward(self, predicts, batch):
|
||||
predict = predicts["predict"]
|
||||
word_predict = predicts["word_out"]
|
||||
gsrm_predict = predicts["gsrm_out"]
|
||||
label = batch[1]
|
||||
|
||||
casted_label = paddle.cast(x=label, dtype="int64")
|
||||
casted_label = paddle.reshape(x=casted_label, shape=[-1, 1])
|
||||
|
||||
cost_word = self.loss_func(word_predict, label=casted_label)
|
||||
cost_gsrm = self.loss_func(gsrm_predict, label=casted_label)
|
||||
cost_vsfd = self.loss_func(predict, label=casted_label)
|
||||
|
||||
cost_word = paddle.reshape(x=paddle.sum(cost_word), shape=[1])
|
||||
cost_gsrm = paddle.reshape(x=paddle.sum(cost_gsrm), shape=[1])
|
||||
cost_vsfd = paddle.reshape(x=paddle.sum(cost_vsfd), shape=[1])
|
||||
|
||||
sum_cost = cost_word * 3.0 + cost_vsfd + cost_gsrm * 0.15
|
||||
|
||||
return {"loss": sum_cost, "word_loss": cost_word, "img_loss": cost_vsfd}
|
||||
54
ppocr/losses/rec_unimernet_loss.py
Normal file
54
ppocr/losses/rec_unimernet_loss.py
Normal file
@@ -0,0 +1,54 @@
|
||||
# copyright (c) 2024 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import paddle
|
||||
import paddle.nn as nn
|
||||
import paddle.nn.functional as F
|
||||
import numpy as np
|
||||
|
||||
|
||||
class UniMERNetLoss(nn.Layer):
|
||||
def __init__(self, length_aware=True, vocab_size=50000):
|
||||
super(UniMERNetLoss, self).__init__()
|
||||
self.ignore_index = -100
|
||||
self.vocab_size = vocab_size
|
||||
self.pad_token_id = 1
|
||||
self.length_aware = length_aware
|
||||
self.cross = nn.CrossEntropyLoss(
|
||||
reduction="mean", ignore_index=self.ignore_index
|
||||
)
|
||||
self.counting_loss_fct = nn.SmoothL1Loss()
|
||||
|
||||
def _get_count_gt(self, labels):
|
||||
mask = (labels != self.pad_token_id).cast("float32")
|
||||
one_hot_labels = F.one_hot(
|
||||
labels, num_classes=self.vocab_size
|
||||
) * mask.unsqueeze(-1)
|
||||
count_gt = paddle.sum(one_hot_labels, axis=1)
|
||||
return count_gt
|
||||
|
||||
def forward(self, preds, batch):
|
||||
logits, count_pred, masked_label = preds
|
||||
labels = batch[1][:, 1:]
|
||||
word_loss = self.cross(
|
||||
paddle.reshape(logits, [-1, logits.shape[-1]]),
|
||||
paddle.reshape(masked_label[:, 1:], [-1]),
|
||||
)
|
||||
loss = word_loss
|
||||
if self.length_aware:
|
||||
count_gt = self._get_count_gt(labels)
|
||||
count_gt = paddle.log(count_gt.cast(paddle.float32) + 1)
|
||||
count_loss = self.counting_loss_fct(count_pred, count_gt)
|
||||
loss += 0.5 * count_loss
|
||||
return {"loss": loss, "word_loss": word_loss, "count_loss": count_loss}
|
||||
70
ppocr/losses/rec_vl_loss.py
Normal file
70
ppocr/losses/rec_vl_loss.py
Normal file
@@ -0,0 +1,70 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/wangyuxin87/VisionLAN
|
||||
"""
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
|
||||
|
||||
class VLLoss(nn.Layer):
|
||||
def __init__(self, mode="LF_1", weight_res=0.5, weight_mas=0.5, **kwargs):
|
||||
super(VLLoss, self).__init__()
|
||||
self.loss_func = paddle.nn.loss.CrossEntropyLoss(reduction="mean")
|
||||
assert mode in ["LF_1", "LF_2", "LA"]
|
||||
self.mode = mode
|
||||
self.weight_res = weight_res
|
||||
self.weight_mas = weight_mas
|
||||
|
||||
def flatten_label(self, target):
|
||||
label_flatten = []
|
||||
label_length = []
|
||||
for i in range(0, target.shape[0]):
|
||||
cur_label = target[i].tolist()
|
||||
label_flatten += cur_label[: cur_label.index(0) + 1]
|
||||
label_length.append(cur_label.index(0) + 1)
|
||||
label_flatten = paddle.to_tensor(label_flatten, dtype="int64")
|
||||
label_length = paddle.to_tensor(label_length, dtype="int32")
|
||||
return (label_flatten, label_length)
|
||||
|
||||
def _flatten(self, sources, lengths):
|
||||
return paddle.concat([t[:l] for t, l in zip(sources, lengths)])
|
||||
|
||||
def forward(self, predicts, batch):
|
||||
text_pre = predicts[0]
|
||||
target = batch[1].astype("int64")
|
||||
label_flatten, length = self.flatten_label(target)
|
||||
text_pre = self._flatten(text_pre, length)
|
||||
if self.mode == "LF_1":
|
||||
loss = self.loss_func(text_pre, label_flatten)
|
||||
else:
|
||||
text_rem = predicts[1]
|
||||
text_mas = predicts[2]
|
||||
target_res = batch[2].astype("int64")
|
||||
target_sub = batch[3].astype("int64")
|
||||
label_flatten_res, length_res = self.flatten_label(target_res)
|
||||
label_flatten_sub, length_sub = self.flatten_label(target_sub)
|
||||
text_rem = self._flatten(text_rem, length_res)
|
||||
text_mas = self._flatten(text_mas, length_sub)
|
||||
loss_ori = self.loss_func(text_pre, label_flatten)
|
||||
loss_res = self.loss_func(text_rem, label_flatten_res)
|
||||
loss_mas = self.loss_func(text_mas, label_flatten_sub)
|
||||
loss = loss_ori + loss_res * self.weight_res + loss_mas * self.weight_mas
|
||||
return {"loss": loss}
|
||||
63
ppocr/losses/stroke_focus_loss.py
Normal file
63
ppocr/losses/stroke_focus_loss.py
Normal file
@@ -0,0 +1,63 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/FudanVI/FudanOCR/blob/main/text-gestalt/loss/stroke_focus_loss.py
|
||||
"""
|
||||
import cv2
|
||||
import sys
|
||||
import time
|
||||
import string
|
||||
import random
|
||||
import numpy as np
|
||||
import paddle.nn as nn
|
||||
import paddle
|
||||
|
||||
|
||||
class StrokeFocusLoss(nn.Layer):
|
||||
def __init__(self, character_dict_path=None, **kwargs):
|
||||
super(StrokeFocusLoss, self).__init__(character_dict_path)
|
||||
self.mse_loss = nn.MSELoss()
|
||||
self.ce_loss = nn.CrossEntropyLoss()
|
||||
self.l1_loss = nn.L1Loss()
|
||||
self.english_stroke_alphabet = "0123456789"
|
||||
self.english_stroke_dict = {}
|
||||
for index in range(len(self.english_stroke_alphabet)):
|
||||
self.english_stroke_dict[self.english_stroke_alphabet[index]] = index
|
||||
|
||||
stroke_decompose_lines = open(character_dict_path, "r").readlines()
|
||||
self.dic = {}
|
||||
for line in stroke_decompose_lines:
|
||||
line = line.strip()
|
||||
character, sequence = line.split()
|
||||
self.dic[character] = sequence
|
||||
|
||||
def forward(self, pred, data):
|
||||
sr_img = pred["sr_img"]
|
||||
hr_img = pred["hr_img"]
|
||||
|
||||
mse_loss = self.mse_loss(sr_img, hr_img)
|
||||
word_attention_map_gt = pred["word_attention_map_gt"]
|
||||
word_attention_map_pred = pred["word_attention_map_pred"]
|
||||
|
||||
hr_pred = pred["hr_pred"]
|
||||
sr_pred = pred["sr_pred"]
|
||||
|
||||
attention_loss = paddle.nn.functional.l1_loss(
|
||||
word_attention_map_gt, word_attention_map_pred
|
||||
)
|
||||
|
||||
loss = (mse_loss + attention_loss * 50) * 100
|
||||
|
||||
return {"mse_loss": mse_loss, "attention_loss": attention_loss, "loss": loss}
|
||||
100
ppocr/losses/table_att_loss.py
Normal file
100
ppocr/losses/table_att_loss.py
Normal file
@@ -0,0 +1,100 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
from paddle.nn import functional as F
|
||||
|
||||
|
||||
class TableAttentionLoss(nn.Layer):
|
||||
def __init__(self, structure_weight=1.0, loc_weight=0.0, **kwargs):
|
||||
super(TableAttentionLoss, self).__init__()
|
||||
self.loss_func = nn.CrossEntropyLoss(weight=None, reduction="none")
|
||||
self.structure_weight = structure_weight
|
||||
self.loc_weight = loc_weight
|
||||
|
||||
def forward(self, predicts, batch):
|
||||
structure_probs = predicts["structure_probs"]
|
||||
structure_targets = batch[1].astype("int64")
|
||||
structure_targets = structure_targets[:, 1:]
|
||||
structure_probs = paddle.reshape(
|
||||
structure_probs, [-1, structure_probs.shape[-1]]
|
||||
)
|
||||
structure_targets = paddle.reshape(structure_targets, [-1])
|
||||
structure_loss = self.loss_func(structure_probs, structure_targets)
|
||||
|
||||
structure_loss = paddle.mean(structure_loss) * self.structure_weight
|
||||
|
||||
loc_preds = predicts["loc_preds"]
|
||||
loc_targets = batch[2].astype("float32")
|
||||
loc_targets_mask = batch[3].astype("float32")
|
||||
loc_targets = loc_targets[:, 1:, :]
|
||||
loc_targets_mask = loc_targets_mask[:, 1:, :]
|
||||
loc_loss = (
|
||||
F.mse_loss(loc_preds * loc_targets_mask, loc_targets) * self.loc_weight
|
||||
)
|
||||
|
||||
total_loss = structure_loss + loc_loss
|
||||
return {
|
||||
"loss": total_loss,
|
||||
"structure_loss": structure_loss,
|
||||
"loc_loss": loc_loss,
|
||||
}
|
||||
|
||||
|
||||
class SLALoss(nn.Layer):
|
||||
def __init__(self, structure_weight=1.0, loc_weight=0.0, loc_loss="mse", **kwargs):
|
||||
super(SLALoss, self).__init__()
|
||||
self.loss_func = nn.CrossEntropyLoss(weight=None, reduction="mean")
|
||||
self.structure_weight = structure_weight
|
||||
self.loc_weight = loc_weight
|
||||
self.loc_loss = loc_loss
|
||||
self.eps = 1e-12
|
||||
|
||||
def forward(self, predicts, batch):
|
||||
structure_probs = predicts["structure_probs"]
|
||||
structure_targets = batch[1].astype("int64")
|
||||
max_len = batch[-2].max().astype("int32")
|
||||
structure_targets = structure_targets[:, 1 : max_len + 2]
|
||||
|
||||
structure_loss = self.loss_func(structure_probs, structure_targets)
|
||||
|
||||
structure_loss = paddle.mean(structure_loss) * self.structure_weight
|
||||
|
||||
loc_preds = predicts["loc_preds"]
|
||||
loc_targets = batch[2].astype("float32")
|
||||
loc_targets_mask = batch[3].astype("float32")
|
||||
loc_targets = loc_targets[:, 1 : max_len + 2]
|
||||
loc_targets_mask = loc_targets_mask[:, 1 : max_len + 2]
|
||||
|
||||
loc_loss = (
|
||||
F.smooth_l1_loss(
|
||||
loc_preds * loc_targets_mask,
|
||||
loc_targets * loc_targets_mask,
|
||||
reduction="sum",
|
||||
)
|
||||
* self.loc_weight
|
||||
)
|
||||
|
||||
loc_loss = loc_loss / (loc_targets_mask.sum() + self.eps)
|
||||
total_loss = structure_loss + loc_loss
|
||||
return {
|
||||
"loss": total_loss,
|
||||
"structure_loss": structure_loss,
|
||||
"loc_loss": loc_loss,
|
||||
}
|
||||
74
ppocr/losses/table_master_loss.py
Normal file
74
ppocr/losses/table_master_loss.py
Normal file
@@ -0,0 +1,74 @@
|
||||
# Copyright (c) 2021 PaddlePaddle Authors. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/JiaquanYe/TableMASTER-mmocr/tree/master/mmocr/models/textrecog/losses
|
||||
"""
|
||||
|
||||
import paddle
|
||||
from paddle import nn
|
||||
|
||||
|
||||
class TableMasterLoss(nn.Layer):
|
||||
def __init__(self, ignore_index=-1):
|
||||
super(TableMasterLoss, self).__init__()
|
||||
self.structure_loss = nn.CrossEntropyLoss(
|
||||
ignore_index=ignore_index, reduction="mean"
|
||||
)
|
||||
self.box_loss = nn.L1Loss(reduction="sum")
|
||||
self.eps = 1e-12
|
||||
|
||||
def forward(self, predicts, batch):
|
||||
# structure_loss
|
||||
structure_probs = predicts["structure_probs"]
|
||||
structure_targets = batch[1]
|
||||
structure_targets = structure_targets[:, 1:]
|
||||
structure_probs = structure_probs.reshape([-1, structure_probs.shape[-1]])
|
||||
structure_targets = structure_targets.reshape([-1])
|
||||
|
||||
structure_loss = self.structure_loss(structure_probs, structure_targets)
|
||||
structure_loss = structure_loss.mean()
|
||||
losses = dict(structure_loss=structure_loss)
|
||||
|
||||
# box loss
|
||||
bboxes_preds = predicts["loc_preds"]
|
||||
bboxes_targets = batch[2][:, 1:, :]
|
||||
bbox_masks = batch[3][:, 1:]
|
||||
# mask empty-bbox or non-bbox structure token's bbox.
|
||||
|
||||
masked_bboxes_preds = bboxes_preds * bbox_masks
|
||||
masked_bboxes_targets = bboxes_targets * bbox_masks
|
||||
|
||||
# horizon loss (x and width)
|
||||
horizon_sum_loss = self.box_loss(
|
||||
masked_bboxes_preds[:, :, 0::2], masked_bboxes_targets[:, :, 0::2]
|
||||
)
|
||||
horizon_loss = horizon_sum_loss / (bbox_masks.sum() + self.eps)
|
||||
# vertical loss (y and height)
|
||||
vertical_sum_loss = self.box_loss(
|
||||
masked_bboxes_preds[:, :, 1::2], masked_bboxes_targets[:, :, 1::2]
|
||||
)
|
||||
vertical_loss = vertical_sum_loss / (bbox_masks.sum() + self.eps)
|
||||
|
||||
horizon_loss = horizon_loss.mean()
|
||||
vertical_loss = vertical_loss.mean()
|
||||
all_loss = structure_loss + horizon_loss + vertical_loss
|
||||
losses.update(
|
||||
{
|
||||
"loss": all_loss,
|
||||
"horizon_bbox_loss": horizon_loss,
|
||||
"vertical_bbox_loss": vertical_loss,
|
||||
}
|
||||
)
|
||||
return losses
|
||||
91
ppocr/losses/text_focus_loss.py
Normal file
91
ppocr/losses/text_focus_loss.py
Normal file
@@ -0,0 +1,91 @@
|
||||
# copyright (c) 2022 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""
|
||||
This code is refer from:
|
||||
https://github.com/FudanVI/FudanOCR/blob/main/scene-text-telescope/loss/text_focus_loss.py
|
||||
"""
|
||||
|
||||
import paddle.nn as nn
|
||||
import paddle
|
||||
import numpy as np
|
||||
import pickle as pkl
|
||||
|
||||
standard_alphebet = "-0123456789abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ"
|
||||
standard_dict = {}
|
||||
for index in range(len(standard_alphebet)):
|
||||
standard_dict[standard_alphebet[index]] = index
|
||||
|
||||
|
||||
def load_confuse_matrix(confuse_dict_path):
|
||||
f = open(confuse_dict_path, "rb")
|
||||
data = pkl.load(f)
|
||||
f.close()
|
||||
number = data[:10]
|
||||
upper = data[10:36]
|
||||
lower = data[36:]
|
||||
end = np.ones((1, 62))
|
||||
pad = np.ones((63, 1))
|
||||
rearrange_data = np.concatenate((end, number, lower, upper), axis=0)
|
||||
rearrange_data = np.concatenate((pad, rearrange_data), axis=1)
|
||||
rearrange_data = 1 / rearrange_data
|
||||
rearrange_data[rearrange_data == np.inf] = 1
|
||||
rearrange_data = paddle.to_tensor(rearrange_data)
|
||||
|
||||
lower_alpha = "abcdefghijklmnopqrstuvwxyz"
|
||||
# upper_alpha = 'ABCDEFGHIJKLMNOPQRSTUVWXYZ'
|
||||
for i in range(63):
|
||||
for j in range(63):
|
||||
if i != j and standard_alphebet[j] in lower_alpha:
|
||||
rearrange_data[i][j] = max(
|
||||
rearrange_data[i][j], rearrange_data[i][j + 26]
|
||||
)
|
||||
rearrange_data = rearrange_data[:37, :37]
|
||||
|
||||
return rearrange_data
|
||||
|
||||
|
||||
def weight_cross_entropy(pred, gt, weight_table):
|
||||
batch = gt.shape[0]
|
||||
weight = weight_table[gt]
|
||||
pred_exp = paddle.exp(pred)
|
||||
pred_exp_weight = weight * pred_exp
|
||||
loss = 0
|
||||
for i in range(len(gt)):
|
||||
loss -= paddle.log(
|
||||
pred_exp_weight[i][gt[i]] / paddle.sum(pred_exp_weight, 1)[i]
|
||||
)
|
||||
return loss / batch
|
||||
|
||||
|
||||
class TelescopeLoss(nn.Layer):
|
||||
def __init__(self, confuse_dict_path):
|
||||
super(TelescopeLoss, self).__init__()
|
||||
self.weight_table = load_confuse_matrix(confuse_dict_path)
|
||||
self.mse_loss = nn.MSELoss()
|
||||
self.ce_loss = nn.CrossEntropyLoss()
|
||||
self.l1_loss = nn.L1Loss()
|
||||
|
||||
def forward(self, pred, data):
|
||||
sr_img = pred["sr_img"]
|
||||
hr_img = pred["hr_img"]
|
||||
sr_pred = pred["sr_pred"]
|
||||
text_gt = pred["text_gt"]
|
||||
|
||||
word_attention_map_gt = pred["word_attention_map_gt"]
|
||||
word_attention_map_pred = pred["word_attention_map_pred"]
|
||||
mse_loss = self.mse_loss(sr_img, hr_img)
|
||||
attention_loss = self.l1_loss(word_attention_map_gt, word_attention_map_pred)
|
||||
recognition_loss = weight_cross_entropy(sr_pred, text_gt, self.weight_table)
|
||||
loss = mse_loss + attention_loss * 10 + recognition_loss * 0.0005
|
||||
return {"mse_loss": mse_loss, "attention_loss": attention_loss, "loss": loss}
|
||||
61
ppocr/losses/vqa_token_layoutlm_loss.py
Executable file
61
ppocr/losses/vqa_token_layoutlm_loss.py
Executable file
@@ -0,0 +1,61 @@
|
||||
# copyright (c) 2021 PaddlePaddle Authors. All Rights Reserve.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from __future__ import absolute_import
|
||||
from __future__ import division
|
||||
from __future__ import print_function
|
||||
|
||||
from paddle import nn
|
||||
from ppocr.losses.basic_loss import DMLLoss
|
||||
|
||||
|
||||
class VQASerTokenLayoutLMLoss(nn.Layer):
|
||||
def __init__(self, num_classes, key=None):
|
||||
super().__init__()
|
||||
self.loss_class = nn.CrossEntropyLoss()
|
||||
self.num_classes = num_classes
|
||||
self.ignore_index = self.loss_class.ignore_index
|
||||
self.key = key
|
||||
|
||||
def forward(self, predicts, batch):
|
||||
if isinstance(predicts, dict) and self.key is not None:
|
||||
predicts = predicts[self.key]
|
||||
labels = batch[5]
|
||||
attention_mask = batch[2]
|
||||
if attention_mask is not None:
|
||||
active_loss = (
|
||||
attention_mask.reshape(
|
||||
[
|
||||
-1,
|
||||
]
|
||||
)
|
||||
== 1
|
||||
)
|
||||
active_output = predicts.reshape([-1, self.num_classes])[active_loss]
|
||||
active_label = labels.reshape(
|
||||
[
|
||||
-1,
|
||||
]
|
||||
)[active_loss]
|
||||
loss = self.loss_class(active_output, active_label)
|
||||
else:
|
||||
loss = self.loss_class(
|
||||
predicts.reshape([-1, self.num_classes]),
|
||||
labels.reshape(
|
||||
[
|
||||
-1,
|
||||
]
|
||||
),
|
||||
)
|
||||
return {"loss": loss}
|
||||
Reference in New Issue
Block a user