lint code (#970)

This commit is contained in:
tastelikefeet
2024-09-02 10:16:35 +08:00
committed by GitHub
parent ce1ee32fde
commit 96e33878b4
48 changed files with 1660 additions and 1362 deletions

View File

@@ -39,7 +39,7 @@ class Models(object):
body_3d_keypoints_hdformer = 'hdformer'
crowd_counting = 'HRNetCrowdCounting'
face_2d_keypoints = 'face-2d-keypoints'
star_68ldk_detection = 'star-68ldk-detection'
star_68ldk_detection = 'star-68ldk-detection'
panoptic_segmentation = 'swinL-panoptic-segmentation'
r50_panoptic_segmentation = 'r50-panoptic-segmentation'
image_reid_person = 'passvitb'

View File

@@ -1 +1 @@
from .alignment import Alignment
from .alignment import Alignment

View File

@@ -1,239 +1,353 @@
import os.path as osp
from .base import Base
class Alignment(Base):
"""
Alignment configure file, which contains training parameters of alignment.
"""
def __init__(self, args):
super(Alignment, self).__init__('alignment')
self.ckpt_dir = '/mnt/workspace/humanAIGC/project/STAR/weights'
self.net = "stackedHGnet_v1"
self.nstack = 4
self.loader_type = "alignment"
self.data_definition = "300W" # COFW, 300W, WFLW
self.test_file = "test.tsv"
# image
self.channels = 3
self.width = 256
self.height = 256
self.means = (127.5, 127.5, 127.5)
self.scale = 1 / 127.5
self.aug_prob = 1.0
self.display_iteration = 10
self.val_epoch = 1
self.valset = "test.tsv"
self.norm_type = 'default'
self.encoder_type = 'default'
self.decoder_type = 'default'
# scheduler & optimizer
self.milestones = [200, 350, 450]
self.max_epoch = 260
self.optimizer = "adam"
self.learn_rate = 0.001
self.weight_decay = 0.00001
self.betas = [0.9, 0.999]
self.gamma = 0.1
# batch_size & workers
self.batch_size = 32
self.train_num_workers = 16
self.val_batch_size = 32
self.val_num_workers = 16
self.test_batch_size = 16
self.test_num_workers = 0
# tricks
self.ema = True
self.add_coord = True
self.use_AAM = True
# loss
self.loss_func = "STARLoss_v2"
# STAR Loss paras
self.star_w = 1
self.star_dist = 'smoothl1'
self.init_from_args(args)
# COFW
if self.data_definition == "COFW":
self.edge_info = (
(True, (0, 4, 2, 5)), # RightEyebrow
(True, (1, 6, 3, 7)), # LeftEyebrow
(True, (8, 12, 10, 13)), # RightEye
(False, (9, 14, 11, 15)), # LeftEye
(True, (18, 20, 19, 21)), # Nose
(True, (22, 26, 23, 27)), # LowerLip
(True, (22, 24, 23, 25)), # UpperLip
)
if self.norm_type == 'ocular':
self.nme_left_index = 8 # ocular
self.nme_right_index = 9 # ocular
elif self.norm_type in ['pupil', 'default']:
self.nme_left_index = 16 # pupil
self.nme_right_index = 17 # pupil
else:
raise NotImplementedError
self.classes_num = [29, 7, 29]
self.crop_op = True
self.flip_mapping = (
[0, 1], [4, 6], [2, 3], [5, 7], [8, 9], [10, 11], [12, 14], [16, 17], [13, 15], [18, 19], [22, 23],
)
self.image_dir = osp.join(self.image_dir, 'COFW')
# 300W
elif self.data_definition == "300W":
self.edge_info = (
(False, (0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16)), # FaceContour
(False, (17, 18, 19, 20, 21)), # RightEyebrow
(False, (22, 23, 24, 25, 26)), # LeftEyebrow
(False, (27, 28, 29, 30)), # NoseLine
(False, (31, 32, 33, 34, 35)), # Nose
(True, (36, 37, 38, 39, 40, 41)), # RightEye
(True, (42, 43, 44, 45, 46, 47)), # LeftEye
(True, (48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 58, 59)), # OuterLip
(True, (60, 61, 62, 63, 64, 65, 66, 67)), # InnerLip
)
if self.norm_type in ['ocular', 'default']:
self.nme_left_index = 36 # ocular
self.nme_right_index = 45 # ocular
elif self.norm_type == 'pupil':
self.nme_left_index = [36, 37, 38, 39, 40, 41] # pupil
self.nme_right_index = [42, 43, 44, 45, 46, 47] # pupil
else:
raise NotImplementedError
self.classes_num = [68, 9, 68]
self.crop_op = True
self.flip_mapping = (
[0, 16], [1, 15], [2, 14], [3, 13], [4, 12], [5, 11], [6, 10], [7, 9],
[17, 26], [18, 25], [19, 24], [20, 23], [21, 22],
[31, 35], [32, 34],
[36, 45], [37, 44], [38, 43], [39, 42], [40, 47], [41, 46],
[48, 54], [49, 53], [50, 52], [61, 63], [60, 64], [67, 65], [58, 56], [59, 55],
)
self.image_dir = osp.join(self.image_dir, '300W')
# self.image_dir = osp.join(self.image_dir, '300VW_images')
# 300VW
elif self.data_definition == "300VW":
self.edge_info = (
(False, (0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16)), # FaceContour
(False, (17, 18, 19, 20, 21)), # RightEyebrow
(False, (22, 23, 24, 25, 26)), # LeftEyebrow
(False, (27, 28, 29, 30)), # NoseLine
(False, (31, 32, 33, 34, 35)), # Nose
(True, (36, 37, 38, 39, 40, 41)), # RightEye
(True, (42, 43, 44, 45, 46, 47)), # LeftEye
(True, (48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 58, 59)), # OuterLip
(True, (60, 61, 62, 63, 64, 65, 66, 67)), # InnerLip
)
if self.norm_type in ['ocular', 'default']:
self.nme_left_index = 36 # ocular
self.nme_right_index = 45 # ocular
elif self.norm_type == 'pupil':
self.nme_left_index = [36, 37, 38, 39, 40, 41] # pupil
self.nme_right_index = [42, 43, 44, 45, 46, 47] # pupil
else:
raise NotImplementedError
self.classes_num = [68, 9, 68]
self.crop_op = True
self.flip_mapping = (
[0, 16], [1, 15], [2, 14], [3, 13], [4, 12], [5, 11], [6, 10], [7, 9],
[17, 26], [18, 25], [19, 24], [20, 23], [21, 22],
[31, 35], [32, 34],
[36, 45], [37, 44], [38, 43], [39, 42], [40, 47], [41, 46],
[48, 54], [49, 53], [50, 52], [61, 63], [60, 64], [67, 65], [58, 56], [59, 55],
)
self.image_dir = osp.join(self.image_dir, '300VW_Dataset_2015_12_14')
# WFLW
elif self.data_definition == "WFLW":
self.edge_info = (
(False, (
0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26,
27,
28, 29, 30, 31, 32)), # FaceContour
(True, (33, 34, 35, 36, 37, 38, 39, 40, 41)), # RightEyebrow
(True, (42, 43, 44, 45, 46, 47, 48, 49, 50)), # LeftEyebrow
(False, (51, 52, 53, 54)), # NoseLine
(False, (55, 56, 57, 58, 59)), # Nose
(True, (60, 61, 62, 63, 64, 65, 66, 67)), # RightEye
(True, (68, 69, 70, 71, 72, 73, 74, 75)), # LeftEye
(True, (76, 77, 78, 79, 80, 81, 82, 83, 84, 85, 86, 87)), # OuterLip
(True, (88, 89, 90, 91, 92, 93, 94, 95)), # InnerLip
)
if self.norm_type in ['ocular', 'default']:
self.nme_left_index = 60 # ocular
self.nme_right_index = 72 # ocular
elif self.norm_type == 'pupil':
self.nme_left_index = 96 # pupils
self.nme_right_index = 97 # pupils
else:
raise NotImplementedError
self.classes_num = [98, 9, 98]
self.crop_op = True
self.flip_mapping = (
[0, 32], [1, 31], [2, 30], [3, 29], [4, 28], [5, 27], [6, 26], [7, 25], [8, 24], [9, 23], [10, 22],
[11, 21], [12, 20], [13, 19], [14, 18], [15, 17], # cheek
[33, 46], [34, 45], [35, 44], [36, 43], [37, 42], [38, 50], [39, 49], [40, 48], [41, 47], # elbrow
[60, 72], [61, 71], [62, 70], [63, 69], [64, 68], [65, 75], [66, 74], [67, 73],
[55, 59], [56, 58],
[76, 82], [77, 81], [78, 80], [87, 83], [86, 84],
[88, 92], [89, 91], [95, 93], [96, 97]
)
self.image_dir = osp.join(self.image_dir, 'WFLW', 'WFLW_images')
self.label_num = self.nstack * 3 if self.use_AAM else self.nstack
self.loss_weights, self.criterions, self.metrics = [], [], []
for i in range(self.nstack):
factor = (2 ** i) / (2 ** (self.nstack - 1))
if self.use_AAM:
self.loss_weights += [factor * weight for weight in [1.0, 10.0, 10.0]]
self.criterions += [self.loss_func, "AWingLoss", "AWingLoss"]
self.metrics += ["NME", None, None]
else:
self.loss_weights += [factor * weight for weight in [1.0]]
self.criterions += [self.loss_func, ]
self.metrics += ["NME", ]
self.key_metric_index = (self.nstack - 1) * 3 if self.use_AAM else (self.nstack - 1)
# data
self.folder = self.get_foldername()
self.work_dir = osp.join(self.ckpt_dir, self.data_definition, self.folder)
self.model_dir = osp.join(self.work_dir, 'model')
self.log_dir = osp.join(self.work_dir, 'log')
self.train_tsv_file = osp.join(self.annot_dir, self.data_definition, "train.tsv")
self.train_pic_dir = self.image_dir
self.val_tsv_file = osp.join(self.annot_dir, self.data_definition, self.valset)
self.val_pic_dir = self.image_dir
self.test_tsv_file = osp.join(self.annot_dir, self.data_definition, self.test_file)
self.test_pic_dir = self.image_dir
# self.train_tsv_file = osp.join(self.annot_dir, '300VW', "train.tsv")
# self.train_pic_dir = self.image_dir
# self.val_tsv_file = osp.join(self.annot_dir, '300VW', self.valset)
# self.val_pic_dir = self.image_dir
# self.test_tsv_file = osp.join(self.annot_dir, '300VW', self.test_file)
# self.test_pic_dir = self.image_dir
def get_foldername(self):
str = ''
str += '{}_{}x{}_{}_ep{}_lr{}_bs{}'.format(self.data_definition, self.height, self.width,
self.optimizer, self.max_epoch, self.learn_rate, self.batch_size)
str += '_{}'.format(self.loss_func)
str += '_{}_{}'.format(self.star_dist, self.star_w) if self.loss_func == 'STARLoss' else ''
str += '_AAM' if self.use_AAM else ''
str += '_{}'.format(self.valset[:-4]) if self.valset != 'test.tsv' else ''
str += '_{}'.format(self.id)
return str
import os.path as osp
from .base import Base
class Alignment(Base):
"""
Alignment configure file, which contains training parameters of alignment.
"""
def __init__(self, args):
super(Alignment, self).__init__('alignment')
self.ckpt_dir = '/mnt/workspace/humanAIGC/project/STAR/weights'
self.net = 'stackedHGnet_v1'
self.nstack = 4
self.loader_type = 'alignment'
self.data_definition = '300W' # COFW, 300W, WFLW
self.test_file = 'test.tsv'
# image
self.channels = 3
self.width = 256
self.height = 256
self.means = (127.5, 127.5, 127.5)
self.scale = 1 / 127.5
self.aug_prob = 1.0
self.display_iteration = 10
self.val_epoch = 1
self.valset = 'test.tsv'
self.norm_type = 'default'
self.encoder_type = 'default'
self.decoder_type = 'default'
# scheduler & optimizer
self.milestones = [200, 350, 450]
self.max_epoch = 260
self.optimizer = 'adam'
self.learn_rate = 0.001
self.weight_decay = 0.00001
self.betas = [0.9, 0.999]
self.gamma = 0.1
# batch_size & workers
self.batch_size = 32
self.train_num_workers = 16
self.val_batch_size = 32
self.val_num_workers = 16
self.test_batch_size = 16
self.test_num_workers = 0
# tricks
self.ema = True
self.add_coord = True
self.use_AAM = True
# loss
self.loss_func = 'STARLoss_v2'
# STAR Loss paras
self.star_w = 1
self.star_dist = 'smoothl1'
self.init_from_args(args)
# COFW
if self.data_definition == 'COFW':
self.edge_info = (
(True, (0, 4, 2, 5)), # RightEyebrow
(True, (1, 6, 3, 7)), # LeftEyebrow
(True, (8, 12, 10, 13)), # RightEye
(False, (9, 14, 11, 15)), # LeftEye
(True, (18, 20, 19, 21)), # Nose
(True, (22, 26, 23, 27)), # LowerLip
(True, (22, 24, 23, 25)), # UpperLip
)
if self.norm_type == 'ocular':
self.nme_left_index = 8 # ocular
self.nme_right_index = 9 # ocular
elif self.norm_type in ['pupil', 'default']:
self.nme_left_index = 16 # pupil
self.nme_right_index = 17 # pupil
else:
raise NotImplementedError
self.classes_num = [29, 7, 29]
self.crop_op = True
self.flip_mapping = (
[0, 1],
[4, 6],
[2, 3],
[5, 7],
[8, 9],
[10, 11],
[12, 14],
[16, 17],
[13, 15],
[18, 19],
[22, 23],
)
self.image_dir = osp.join(self.image_dir, 'COFW')
# 300W
elif self.data_definition == '300W':
self.edge_info = (
(False, (0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15,
16)), # FaceContour
(False, (17, 18, 19, 20, 21)), # RightEyebrow
(False, (22, 23, 24, 25, 26)), # LeftEyebrow
(False, (27, 28, 29, 30)), # NoseLine
(False, (31, 32, 33, 34, 35)), # Nose
(True, (36, 37, 38, 39, 40, 41)), # RightEye
(True, (42, 43, 44, 45, 46, 47)), # LeftEye
(True, (48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 58,
59)), # OuterLip
(True, (60, 61, 62, 63, 64, 65, 66, 67)), # InnerLip
)
if self.norm_type in ['ocular', 'default']:
self.nme_left_index = 36 # ocular
self.nme_right_index = 45 # ocular
elif self.norm_type == 'pupil':
self.nme_left_index = [36, 37, 38, 39, 40, 41] # pupil
self.nme_right_index = [42, 43, 44, 45, 46, 47] # pupil
else:
raise NotImplementedError
self.classes_num = [68, 9, 68]
self.crop_op = True
self.flip_mapping = (
[0, 16],
[1, 15],
[2, 14],
[3, 13],
[4, 12],
[5, 11],
[6, 10],
[7, 9],
[17, 26],
[18, 25],
[19, 24],
[20, 23],
[21, 22],
[31, 35],
[32, 34],
[36, 45],
[37, 44],
[38, 43],
[39, 42],
[40, 47],
[41, 46],
[48, 54],
[49, 53],
[50, 52],
[61, 63],
[60, 64],
[67, 65],
[58, 56],
[59, 55],
)
self.image_dir = osp.join(self.image_dir, '300W')
# self.image_dir = osp.join(self.image_dir, '300VW_images')
# 300VW
elif self.data_definition == '300VW':
self.edge_info = (
(False, (0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15,
16)), # FaceContour
(False, (17, 18, 19, 20, 21)), # RightEyebrow
(False, (22, 23, 24, 25, 26)), # LeftEyebrow
(False, (27, 28, 29, 30)), # NoseLine
(False, (31, 32, 33, 34, 35)), # Nose
(True, (36, 37, 38, 39, 40, 41)), # RightEye
(True, (42, 43, 44, 45, 46, 47)), # LeftEye
(True, (48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 58,
59)), # OuterLip
(True, (60, 61, 62, 63, 64, 65, 66, 67)), # InnerLip
)
if self.norm_type in ['ocular', 'default']:
self.nme_left_index = 36 # ocular
self.nme_right_index = 45 # ocular
elif self.norm_type == 'pupil':
self.nme_left_index = [36, 37, 38, 39, 40, 41] # pupil
self.nme_right_index = [42, 43, 44, 45, 46, 47] # pupil
else:
raise NotImplementedError
self.classes_num = [68, 9, 68]
self.crop_op = True
self.flip_mapping = (
[0, 16],
[1, 15],
[2, 14],
[3, 13],
[4, 12],
[5, 11],
[6, 10],
[7, 9],
[17, 26],
[18, 25],
[19, 24],
[20, 23],
[21, 22],
[31, 35],
[32, 34],
[36, 45],
[37, 44],
[38, 43],
[39, 42],
[40, 47],
[41, 46],
[48, 54],
[49, 53],
[50, 52],
[61, 63],
[60, 64],
[67, 65],
[58, 56],
[59, 55],
)
self.image_dir = osp.join(self.image_dir,
'300VW_Dataset_2015_12_14')
# WFLW
elif self.data_definition == 'WFLW':
self.edge_info = (
(False, (0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15,
16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28,
29, 30, 31, 32)), # FaceContour
(True, (33, 34, 35, 36, 37, 38, 39, 40, 41)), # RightEyebrow
(True, (42, 43, 44, 45, 46, 47, 48, 49, 50)), # LeftEyebrow
(False, (51, 52, 53, 54)), # NoseLine
(False, (55, 56, 57, 58, 59)), # Nose
(True, (60, 61, 62, 63, 64, 65, 66, 67)), # RightEye
(True, (68, 69, 70, 71, 72, 73, 74, 75)), # LeftEye
(True, (76, 77, 78, 79, 80, 81, 82, 83, 84, 85, 86,
87)), # OuterLip
(True, (88, 89, 90, 91, 92, 93, 94, 95)), # InnerLip
)
if self.norm_type in ['ocular', 'default']:
self.nme_left_index = 60 # ocular
self.nme_right_index = 72 # ocular
elif self.norm_type == 'pupil':
self.nme_left_index = 96 # pupils
self.nme_right_index = 97 # pupils
else:
raise NotImplementedError
self.classes_num = [98, 9, 98]
self.crop_op = True
self.flip_mapping = (
[0, 32],
[1, 31],
[2, 30],
[3, 29],
[4, 28],
[5, 27],
[6, 26],
[7, 25],
[8, 24],
[9, 23],
[10, 22],
[11, 21],
[12, 20],
[13, 19],
[14, 18],
[15, 17], # cheek
[33, 46],
[34, 45],
[35, 44],
[36, 43],
[37, 42],
[38, 50],
[39, 49],
[40, 48],
[41, 47], # elbrow
[60, 72],
[61, 71],
[62, 70],
[63, 69],
[64, 68],
[65, 75],
[66, 74],
[67, 73],
[55, 59],
[56, 58],
[76, 82],
[77, 81],
[78, 80],
[87, 83],
[86, 84],
[88, 92],
[89, 91],
[95, 93],
[96, 97])
self.image_dir = osp.join(self.image_dir, 'WFLW', 'WFLW_images')
self.label_num = self.nstack * 3 if self.use_AAM else self.nstack
self.loss_weights, self.criterions, self.metrics = [], [], []
for i in range(self.nstack):
factor = (2**i) / (2**(self.nstack - 1))
if self.use_AAM:
self.loss_weights += [
factor * weight for weight in [1.0, 10.0, 10.0]
]
self.criterions += [self.loss_func, 'AWingLoss', 'AWingLoss']
self.metrics += ['NME', None, None]
else:
self.loss_weights += [factor * weight for weight in [1.0]]
self.criterions += [
self.loss_func,
]
self.metrics += [
'NME',
]
self.key_metric_index = (self.nstack - 1) * 3 if self.use_AAM else (
self.nstack - 1)
# data
self.folder = self.get_foldername()
self.work_dir = osp.join(self.ckpt_dir, self.data_definition,
self.folder)
self.model_dir = osp.join(self.work_dir, 'model')
self.log_dir = osp.join(self.work_dir, 'log')
self.train_tsv_file = osp.join(self.annot_dir, self.data_definition,
'train.tsv')
self.train_pic_dir = self.image_dir
self.val_tsv_file = osp.join(self.annot_dir, self.data_definition,
self.valset)
self.val_pic_dir = self.image_dir
self.test_tsv_file = osp.join(self.annot_dir, self.data_definition,
self.test_file)
self.test_pic_dir = self.image_dir
# self.train_tsv_file = osp.join(self.annot_dir, '300VW', "train.tsv")
# self.train_pic_dir = self.image_dir
# self.val_tsv_file = osp.join(self.annot_dir, '300VW', self.valset)
# self.val_pic_dir = self.image_dir
# self.test_tsv_file = osp.join(self.annot_dir, '300VW', self.test_file)
# self.test_pic_dir = self.image_dir
def get_foldername(self):
str = ''
str += '{}_{}x{}_{}_ep{}_lr{}_bs{}'.format(
self.data_definition, self.height, self.width, self.optimizer,
self.max_epoch, self.learn_rate, self.batch_size)
str += '_{}'.format(self.loss_func)
str += '_{}_{}'.format(
self.star_dist,
self.star_w) if self.loss_func == 'STARLoss' else ''
str += '_AAM' if self.use_AAM else ''
str += '_{}'.format(
self.valset[:-4]) if self.valset != 'test.tsv' else ''
str += '_{}'.format(self.id)
return str

View File

@@ -1,94 +1,102 @@
import uuid
import logging
import os.path as osp
from argparse import Namespace
# from tensorboardX import SummaryWriter
class Base:
"""
Base configure file, which contains the basic training parameters and should be inherited by other attribute configure file.
"""
def __init__(self, config_name, ckpt_dir='./', image_dir='./', annot_dir='./'):
self.type = config_name
self.id = str(uuid.uuid4())
self.note = ""
self.ckpt_dir = ckpt_dir
self.image_dir = image_dir
self.annot_dir = annot_dir
self.loader_type = "alignment"
self.loss_func = "STARLoss"
# train
self.batch_size = 128
self.val_batch_size = 1
self.test_batch_size = 32
self.channels = 3
self.width = 256
self.height = 256
# mean values in r, g, b channel.
self.means = (127, 127, 127)
self.scale = 0.0078125
self.display_iteration = 100
self.milestones = [50, 80]
self.max_epoch = 100
self.net = "stackedHGnet_v1"
self.nstack = 4
# ["adam", "sgd"]
self.optimizer = "adam"
self.learn_rate = 0.1
self.momentum = 0.01 # caffe: 0.99
self.weight_decay = 0.0
self.nesterov = False
self.scheduler = "MultiStepLR"
self.gamma = 0.1
self.loss_weights = [1.0]
self.criterions = ["SoftmaxWithLoss"]
self.metrics = ["Accuracy"]
self.key_metric_index = 0
self.classes_num = [1000]
self.label_num = len(self.classes_num)
# model
self.ema = False
self.use_AAM = True
# visualization
self.writer = None
# log file
self.logger = None
def init_instance(self):
# self.writer = SummaryWriter(logdir=self.log_dir, comment=self.type)
log_formatter = logging.Formatter("%(asctime)s %(levelname)-8s: %(message)s")
root_logger = logging.getLogger()
file_handler = logging.FileHandler(osp.join(self.log_dir, "log.txt"))
file_handler.setFormatter(log_formatter)
file_handler.setLevel(logging.NOTSET)
root_logger.addHandler(file_handler)
console_handler = logging.StreamHandler()
console_handler.setFormatter(log_formatter)
console_handler.setLevel(logging.NOTSET)
root_logger.addHandler(console_handler)
root_logger.setLevel(logging.NOTSET)
self.logger = root_logger
def __del__(self):
# tensorboard --logdir self.log_dir
if self.writer is not None:
# self.writer.export_scalars_to_json(self.log_dir + "visual.json")
self.writer.close()
def init_from_args(self, args: Namespace):
args_vars = vars(args)
for key, value in args_vars.items():
if hasattr(self, key) and value is not None:
setattr(self, key, value)
import logging
import os.path as osp
import uuid
from argparse import Namespace
# from tensorboardX import SummaryWriter
class Base:
"""
Base configure file, which contains the basic training parameters
and should be inherited by other attribute configure file.
"""
def __init__(self,
config_name,
ckpt_dir='./',
image_dir='./',
annot_dir='./'):
self.type = config_name
self.id = str(uuid.uuid4())
self.note = ''
self.ckpt_dir = ckpt_dir
self.image_dir = image_dir
self.annot_dir = annot_dir
self.loader_type = 'alignment'
self.loss_func = 'STARLoss'
# train
self.batch_size = 128
self.val_batch_size = 1
self.test_batch_size = 32
self.channels = 3
self.width = 256
self.height = 256
# mean values in r, g, b channel.
self.means = (127, 127, 127)
self.scale = 0.0078125
self.display_iteration = 100
self.milestones = [50, 80]
self.max_epoch = 100
self.net = 'stackedHGnet_v1'
self.nstack = 4
# ["adam", "sgd"]
self.optimizer = 'adam'
self.learn_rate = 0.1
self.momentum = 0.01 # caffe: 0.99
self.weight_decay = 0.0
self.nesterov = False
self.scheduler = 'MultiStepLR'
self.gamma = 0.1
self.loss_weights = [1.0]
self.criterions = ['SoftmaxWithLoss']
self.metrics = ['Accuracy']
self.key_metric_index = 0
self.classes_num = [1000]
self.label_num = len(self.classes_num)
# model
self.ema = False
self.use_AAM = True
# visualization
self.writer = None
# log file
self.logger = None
def init_instance(self):
# self.writer = SummaryWriter(logdir=self.log_dir, comment=self.type)
log_formatter = logging.Formatter(
'%(asctime)s %(levelname)-8s: %(message)s')
root_logger = logging.getLogger()
file_handler = logging.FileHandler(osp.join(self.log_dir, 'log.txt'))
file_handler.setFormatter(log_formatter)
file_handler.setLevel(logging.NOTSET)
root_logger.addHandler(file_handler)
console_handler = logging.StreamHandler()
console_handler.setFormatter(log_formatter)
console_handler.setLevel(logging.NOTSET)
root_logger.addHandler(console_handler)
root_logger.setLevel(logging.NOTSET)
self.logger = root_logger
def __del__(self):
# tensorboard --logdir self.log_dir
if self.writer is not None:
# self.writer.export_scalars_to_json(self.log_dir + "visual.json")
self.writer.close()
def init_from_args(self, args: Namespace):
args_vars = vars(args)
for key, value in args_vars.items():
if hasattr(self, key) and value is not None:
setattr(self, key, value)

View File

@@ -1,13 +1,15 @@
import cv2
import math
import copy
import numpy as np
import argparse
import copy
import math
import cv2
import numpy as np
import torch
# private package
from .lib import utility
class GetCropMatrix():
"""
from_shape -> transform_matrix
@@ -18,7 +20,8 @@ class GetCropMatrix():
self.target_face_scale = target_face_scale
self.align_corners = align_corners
def _compose_rotate_and_scale(self, angle, scale, shift_xy, from_center, to_center):
def _compose_rotate_and_scale(self, angle, scale, shift_xy, from_center,
to_center):
cosv = math.cos(angle)
sinv = math.sin(angle)
@@ -36,11 +39,8 @@ class GetCropMatrix():
b1 = acos
b2 = ty - asin * fx - acos * fy + shift_xy[1]
rot_scale_m = np.array([
[a0, a1, a2],
[b0, b1, b2],
[0.0, 0.0, 1.0]
], np.float32)
rot_scale_m = np.array([[a0, a1, a2], [b0, b1, b2], [0.0, 0.0, 1.0]],
np.float32)
return rot_scale_m
def process(self, scale, center_w, center_h):
@@ -53,7 +53,9 @@ class GetCropMatrix():
scale_mu = self.image_size / (scale * self.target_face_scale * 200.0)
shift_xy_mu = (0, 0)
matrix = self._compose_rotate_and_scale(
rot_mu, scale_mu, shift_xy_mu,
rot_mu,
scale_mu,
shift_xy_mu,
from_center=[center_w, center_h],
to_center=[to_w / 2.0, to_h / 2.0])
return matrix
@@ -69,8 +71,11 @@ class TransformPerspective():
def process(self, image, matrix):
return cv2.warpPerspective(
image, matrix, dsize=(self.image_size, self.image_size),
flags=cv2.INTER_LINEAR, borderValue=0)
image,
matrix,
dsize=(self.image_size, self.image_size),
flags=cv2.INTER_LINEAR,
borderValue=0)
class TransformPoints2D():
@@ -80,63 +85,78 @@ class TransformPoints2D():
def process(self, srcPoints, matrix):
# nx3
desPoints = np.concatenate([srcPoints, np.ones_like(srcPoints[:, [0]])], axis=1)
desPoints = np.concatenate(
[srcPoints, np.ones_like(srcPoints[:, [0]])], axis=1)
desPoints = desPoints @ np.transpose(matrix) # nx3
desPoints = desPoints[:, :2] / desPoints[:, [2, 2]]
return desPoints.astype(srcPoints.dtype)
class Alignment:
def __init__(self, args, model_path, dl_framework, device_ids):
self.input_size = 256
self.target_face_scale = 1.0
self.dl_framework = dl_framework
# model
if self.dl_framework == "pytorch":
if self.dl_framework == 'pytorch':
# conf
self.config = utility.get_config(args)
self.config.device_id = device_ids[0]
# set environment
utility.set_environment(self.config)
net = utility.get_net(self.config)
if device_ids == [-1]:
checkpoint = torch.load(model_path, map_location="cpu")
checkpoint = torch.load(model_path, map_location='cpu')
else:
checkpoint = torch.load(model_path)
net.load_state_dict(checkpoint["net"])
net.load_state_dict(checkpoint['net'])
if self.config.device_id == -1:
net = net.cpu()
else:
net = net.to(self.config.device_id)
net.eval()
self.alignment = net
else:
assert False
self.getCropMatrix = GetCropMatrix(image_size=self.input_size, target_face_scale=self.target_face_scale,
align_corners=True)
self.transformPerspective = TransformPerspective(image_size=self.input_size)
self.getCropMatrix = GetCropMatrix(
image_size=self.input_size,
target_face_scale=self.target_face_scale,
align_corners=True)
self.transformPerspective = TransformPerspective(
image_size=self.input_size)
self.transformPoints2D = TransformPoints2D()
def norm_points(self, points, align_corners=False):
if align_corners:
# [0, SIZE-1] -> [-1, +1]
return points / torch.tensor([self.input_size - 1, self.input_size - 1]).to(points).view(1, 1, 2) * 2 - 1
return points / torch.tensor([
self.input_size - 1, self.input_size - 1
]).to(points).view(1, 1, 2) * 2 - 1
else:
# [-0.5, SIZE-0.5] -> [-1, +1]
return (points * 2 + 1) / torch.tensor([self.input_size, self.input_size]).to(points).view(1, 1, 2) - 1
return (points * 2 + 1) / torch.tensor([
self.input_size, self.input_size
]).to(points).view(1, 1, 2) - 1
def denorm_points(self, points, align_corners=False):
if align_corners:
# [-1, +1] -> [0, SIZE-1]
return (points + 1) / 2 * torch.tensor([self.input_size - 1, self.input_size - 1]).to(points).view(1, 1, 2)
return (points + 1) / 2 * torch.tensor([
self.input_size - 1, self.input_size - 1
]).to(points).view(1, 1, 2)
else:
# [-1, +1] -> [-0.5, SIZE-0.5]
return ((points + 1) * torch.tensor([self.input_size, self.input_size]).to(points).view(1, 1, 2) - 1) / 2
return ((points + 1) * torch.tensor( # noqa
[self.input_size, self.input_size]).to(points).view(1, 1,
2) # noqa
- 1) / 2 # noqa
def preprocess(self, image, scale, center_w, center_h):
matrix = self.getCropMatrix.process(scale, center_w, center_h)
@@ -151,7 +171,7 @@ class Alignment:
input_tensor = input_tensor.cpu()
else:
input_tensor = input_tensor.to(self.config.device_id)
return input_tensor, matrix
def postprocess(self, srcPoints, coeff):
@@ -160,14 +180,17 @@ class Alignment:
# src = matrix * dst
dstPoints = np.zeros(srcPoints.shape, dtype=np.float32)
for i in range(srcPoints.shape[0]):
dstPoints[i][0] = coeff[0][0] * srcPoints[i][0] + coeff[0][1] * srcPoints[i][1] + coeff[0][2]
dstPoints[i][1] = coeff[1][0] * srcPoints[i][0] + coeff[1][1] * srcPoints[i][1] + coeff[1][2]
dstPoints[i][0] = coeff[0][0] * srcPoints[i][0] + coeff[0][
1] * srcPoints[i][1] + coeff[0][2]
dstPoints[i][1] = coeff[1][0] * srcPoints[i][0] + coeff[1][
1] * srcPoints[i][1] + coeff[1][2]
return dstPoints
def analyze(self, image, scale, center_w, center_h):
input_tensor, matrix = self.preprocess(image, scale, center_w, center_h)
input_tensor, matrix = self.preprocess(image, scale, center_w,
center_h)
if self.dl_framework == "pytorch":
if self.dl_framework == 'pytorch':
with torch.no_grad():
output = self.alignment(input_tensor)
landmarks = output[-1][0]
@@ -179,4 +202,3 @@ class Alignment:
landmarks = self.postprocess(landmarks, np.linalg.inv(matrix))
return landmarks

View File

@@ -1,2 +1,2 @@
from .backbone import StackedHGNetV1
from .utility import get_config, get_net
from .backbone import StackedHGNetV1
from .utility import get_config, get_net

View File

@@ -1,5 +1,5 @@
from .stackedHGNetV1 import StackedHGNetV1
__all__ = [
"StackedHGNetV1",
]
from .stackedHGNetV1 import StackedHGNetV1
__all__ = [
'StackedHGNetV1',
]

View File

@@ -1,157 +1,187 @@
import torch
import torch.nn as nn
class AddCoordsTh(nn.Module):
def __init__(self, x_dim, y_dim, with_r=False, with_boundary=False):
super(AddCoordsTh, self).__init__()
self.x_dim = x_dim
self.y_dim = y_dim
self.with_r = with_r
self.with_boundary = with_boundary
def forward(self, input_tensor, heatmap=None):
"""
input_tensor: (batch, c, x_dim, y_dim)
"""
batch_size_tensor = input_tensor.shape[0]
xx_ones = torch.ones([1, self.y_dim], dtype=torch.int32).to(input_tensor)
xx_ones = xx_ones.unsqueeze(-1)
xx_range = torch.arange(self.x_dim, dtype=torch.int32).unsqueeze(0).to(input_tensor)
xx_range = xx_range.unsqueeze(1)
xx_channel = torch.matmul(xx_ones.float(), xx_range.float())
xx_channel = xx_channel.unsqueeze(-1)
yy_ones = torch.ones([1, self.x_dim], dtype=torch.int32).to(input_tensor)
yy_ones = yy_ones.unsqueeze(1)
yy_range = torch.arange(self.y_dim, dtype=torch.int32).unsqueeze(0).to(input_tensor)
yy_range = yy_range.unsqueeze(-1)
yy_channel = torch.matmul(yy_range.float(), yy_ones.float())
yy_channel = yy_channel.unsqueeze(-1)
xx_channel = xx_channel.permute(0, 3, 2, 1)
yy_channel = yy_channel.permute(0, 3, 2, 1)
xx_channel = xx_channel / (self.x_dim - 1)
yy_channel = yy_channel / (self.y_dim - 1)
xx_channel = xx_channel * 2 - 1
yy_channel = yy_channel * 2 - 1
xx_channel = xx_channel.repeat(batch_size_tensor, 1, 1, 1)
yy_channel = yy_channel.repeat(batch_size_tensor, 1, 1, 1)
if self.with_boundary and type(heatmap) != type(None):
boundary_channel = torch.clamp(heatmap[:, -1:, :, :],
0.0, 1.0)
zero_tensor = torch.zeros_like(xx_channel).to(xx_channel)
xx_boundary_channel = torch.where(boundary_channel>0.05,
xx_channel, zero_tensor)
yy_boundary_channel = torch.where(boundary_channel>0.05,
yy_channel, zero_tensor)
ret = torch.cat([input_tensor, xx_channel, yy_channel], dim=1)
if self.with_r:
rr = torch.sqrt(torch.pow(xx_channel, 2) + torch.pow(yy_channel, 2))
rr = rr / torch.max(rr)
ret = torch.cat([ret, rr], dim=1)
if self.with_boundary and type(heatmap) != type(None):
ret = torch.cat([ret, xx_boundary_channel,
yy_boundary_channel], dim=1)
return ret
class CoordConvTh(nn.Module):
"""CoordConv layer as in the paper."""
def __init__(self, x_dim, y_dim, with_r, with_boundary,
in_channels, out_channels, first_one=False, relu=False, bn=False, *args, **kwargs):
super(CoordConvTh, self).__init__()
self.addcoords = AddCoordsTh(x_dim=x_dim, y_dim=y_dim, with_r=with_r,
with_boundary=with_boundary)
in_channels += 2
if with_r:
in_channels += 1
if with_boundary and not first_one:
in_channels += 2
self.conv = nn.Conv2d(in_channels=in_channels, out_channels=out_channels, *args, **kwargs)
self.relu = nn.ReLU() if relu else None
self.bn = nn.BatchNorm2d(out_channels) if bn else None
self.with_boundary = with_boundary
self.first_one = first_one
def forward(self, input_tensor, heatmap=None):
assert (self.with_boundary and not self.first_one) == (heatmap is not None)
ret = self.addcoords(input_tensor, heatmap)
ret = self.conv(ret)
if self.bn is not None:
ret = self.bn(ret)
if self.relu is not None:
ret = self.relu(ret)
return ret
'''
An alternative implementation for PyTorch with auto-infering the x-y dimensions.
'''
class AddCoords(nn.Module):
def __init__(self, with_r=False):
super().__init__()
self.with_r = with_r
def forward(self, input_tensor):
"""
Args:
input_tensor: shape(batch, channel, x_dim, y_dim)
"""
batch_size, _, x_dim, y_dim = input_tensor.size()
xx_channel = torch.arange(x_dim).repeat(1, y_dim, 1).to(input_tensor)
yy_channel = torch.arange(y_dim).repeat(1, x_dim, 1).transpose(1, 2).to(input_tensor)
xx_channel = xx_channel / (x_dim - 1)
yy_channel = yy_channel / (y_dim - 1)
xx_channel = xx_channel * 2 - 1
yy_channel = yy_channel * 2 - 1
xx_channel = xx_channel.repeat(batch_size, 1, 1, 1).transpose(2, 3)
yy_channel = yy_channel.repeat(batch_size, 1, 1, 1).transpose(2, 3)
ret = torch.cat([
input_tensor,
xx_channel.type_as(input_tensor),
yy_channel.type_as(input_tensor)], dim=1)
if self.with_r:
rr = torch.sqrt(torch.pow(xx_channel - 0.5, 2) + torch.pow(yy_channel - 0.5, 2))
ret = torch.cat([ret, rr], dim=1)
return ret
class CoordConv(nn.Module):
def __init__(self, in_channels, out_channels, with_r=False, **kwargs):
super().__init__()
self.addcoords = AddCoords(with_r=with_r)
in_channels += 2
if with_r:
in_channels += 1
self.conv = nn.Conv2d(in_channels, out_channels, **kwargs)
def forward(self, x):
ret = self.addcoords(x)
ret = self.conv(ret)
return ret
import torch
import torch.nn as nn
class AddCoordsTh(nn.Module):
def __init__(self, x_dim, y_dim, with_r=False, with_boundary=False):
super(AddCoordsTh, self).__init__()
self.x_dim = x_dim
self.y_dim = y_dim
self.with_r = with_r
self.with_boundary = with_boundary
def forward(self, input_tensor, heatmap=None):
"""
input_tensor: (batch, c, x_dim, y_dim)
"""
batch_size_tensor = input_tensor.shape[0]
xx_ones = torch.ones([1, self.y_dim],
dtype=torch.int32).to(input_tensor)
xx_ones = xx_ones.unsqueeze(-1)
xx_range = torch.arange(
self.x_dim, dtype=torch.int32).unsqueeze(0).to(input_tensor)
xx_range = xx_range.unsqueeze(1)
xx_channel = torch.matmul(xx_ones.float(), xx_range.float())
xx_channel = xx_channel.unsqueeze(-1)
yy_ones = torch.ones([1, self.x_dim],
dtype=torch.int32).to(input_tensor)
yy_ones = yy_ones.unsqueeze(1)
yy_range = torch.arange(
self.y_dim, dtype=torch.int32).unsqueeze(0).to(input_tensor)
yy_range = yy_range.unsqueeze(-1)
yy_channel = torch.matmul(yy_range.float(), yy_ones.float())
yy_channel = yy_channel.unsqueeze(-1)
xx_channel = xx_channel.permute(0, 3, 2, 1)
yy_channel = yy_channel.permute(0, 3, 2, 1)
xx_channel = xx_channel / (self.x_dim - 1)
yy_channel = yy_channel / (self.y_dim - 1)
xx_channel = xx_channel * 2 - 1
yy_channel = yy_channel * 2 - 1
xx_channel = xx_channel.repeat(batch_size_tensor, 1, 1, 1)
yy_channel = yy_channel.repeat(batch_size_tensor, 1, 1, 1)
if self.with_boundary and heatmap is not None:
boundary_channel = torch.clamp(heatmap[:, -1:, :, :], 0.0, 1.0)
zero_tensor = torch.zeros_like(xx_channel).to(xx_channel)
xx_boundary_channel = torch.where(boundary_channel > 0.05,
xx_channel, zero_tensor)
yy_boundary_channel = torch.where(boundary_channel > 0.05,
yy_channel, zero_tensor)
ret = torch.cat([input_tensor, xx_channel, yy_channel], dim=1)
if self.with_r:
rr = torch.sqrt(
torch.pow(xx_channel, 2) + torch.pow(yy_channel, 2))
rr = rr / torch.max(rr)
ret = torch.cat([ret, rr], dim=1)
if self.with_boundary and heatmap is not None:
ret = torch.cat([ret, xx_boundary_channel, yy_boundary_channel],
dim=1)
return ret
class CoordConvTh(nn.Module):
"""CoordConv layer as in the paper."""
def __init__(self,
x_dim,
y_dim,
with_r,
with_boundary,
in_channels,
out_channels,
first_one=False,
relu=False,
bn=False,
*args,
**kwargs):
super(CoordConvTh, self).__init__()
self.addcoords = AddCoordsTh(
x_dim=x_dim,
y_dim=y_dim,
with_r=with_r,
with_boundary=with_boundary)
in_channels += 2
if with_r:
in_channels += 1
if with_boundary and not first_one:
in_channels += 2
self.conv = nn.Conv2d(
in_channels=in_channels,
out_channels=out_channels,
*args,
**kwargs)
self.relu = nn.ReLU() if relu else None
self.bn = nn.BatchNorm2d(out_channels) if bn else None
self.with_boundary = with_boundary
self.first_one = first_one
def forward(self, input_tensor, heatmap=None):
assert (self.with_boundary and not self.first_one) == (
heatmap is not None)
ret = self.addcoords(input_tensor, heatmap)
ret = self.conv(ret)
if self.bn is not None:
ret = self.bn(ret)
if self.relu is not None:
ret = self.relu(ret)
return ret
'''
An alternative implementation for PyTorch with auto-infering the x-y dimensions.
'''
class AddCoords(nn.Module):
def __init__(self, with_r=False):
super().__init__()
self.with_r = with_r
def forward(self, input_tensor):
"""
Args:
input_tensor: shape(batch, channel, x_dim, y_dim)
"""
batch_size, _, x_dim, y_dim = input_tensor.size()
xx_channel = torch.arange(x_dim).repeat(1, y_dim, 1).to(input_tensor)
yy_channel = torch.arange(y_dim).repeat(1, x_dim, 1).transpose(
1, 2).to(input_tensor)
xx_channel = xx_channel / (x_dim - 1)
yy_channel = yy_channel / (y_dim - 1)
xx_channel = xx_channel * 2 - 1
yy_channel = yy_channel * 2 - 1
xx_channel = xx_channel.repeat(batch_size, 1, 1, 1).transpose(2, 3)
yy_channel = yy_channel.repeat(batch_size, 1, 1, 1).transpose(2, 3)
ret = torch.cat(
[ # noqa
input_tensor, # noqa
xx_channel.type_as(input_tensor), # noqa
yy_channel.type_as(input_tensor) # noqa
], # noqa
dim=1) # noqa
if self.with_r:
rr = torch.sqrt(
torch.pow(xx_channel - 0.5, 2)
+ torch.pow(yy_channel - 0.5, 2))
ret = torch.cat([ret, rr], dim=1)
return ret
class CoordConv(nn.Module):
def __init__(self, in_channels, out_channels, with_r=False, **kwargs):
super().__init__()
self.addcoords = AddCoords(with_r=with_r)
in_channels += 2
if with_r:
in_channels += 1
self.conv = nn.Conv2d(in_channels, out_channels, **kwargs)
def forward(self, x):
ret = self.addcoords(x)
ret = self.conv(ret)
return ret

View File

@@ -1,307 +1,374 @@
import numpy as np
import torch
import torch.nn as nn
import torch.nn.functional as F
from .core.coord_conv import CoordConvTh
from ..dataset import get_decoder
class Activation(nn.Module):
def __init__(self, kind: str = 'relu', channel=None):
super().__init__()
self.kind = kind
if '+' in kind:
norm_str, act_str = kind.split('+')
else:
norm_str, act_str = 'none', kind
self.norm_fn = {
'in': F.instance_norm,
'bn': nn.BatchNorm2d(channel),
'bn_noaffine': nn.BatchNorm2d(channel, affine=False, track_running_stats=True),
'none': None
}[norm_str]
self.act_fn = {
'relu': F.relu,
'softplus': nn.Softplus(),
'exp': torch.exp,
'sigmoid': torch.sigmoid,
'tanh': torch.tanh,
'none': None
}[act_str]
self.channel = channel
def forward(self, x):
if self.norm_fn is not None:
x = self.norm_fn(x)
if self.act_fn is not None:
x = self.act_fn(x)
return x
def extra_repr(self):
return f'kind={self.kind}, channel={self.channel}'
class ConvBlock(nn.Module):
def __init__(self, inp_dim, out_dim, kernel_size=3, stride=1, bn=False, relu=True, groups=1):
super(ConvBlock, self).__init__()
self.inp_dim = inp_dim
self.conv = nn.Conv2d(inp_dim, out_dim, kernel_size,
stride, padding=(kernel_size - 1) // 2, groups=groups, bias=True)
self.relu = None
self.bn = None
if relu:
self.relu = nn.ReLU()
if bn:
self.bn = nn.BatchNorm2d(out_dim)
def forward(self, x):
x = self.conv(x)
if self.bn is not None:
x = self.bn(x)
if self.relu is not None:
x = self.relu(x)
return x
class ResBlock(nn.Module):
def __init__(self, inp_dim, out_dim, mid_dim=None):
super(ResBlock, self).__init__()
if mid_dim is None:
mid_dim = out_dim // 2
self.relu = nn.ReLU()
self.bn1 = nn.BatchNorm2d(inp_dim)
self.conv1 = ConvBlock(inp_dim, mid_dim, 1, relu=False)
self.bn2 = nn.BatchNorm2d(mid_dim)
self.conv2 = ConvBlock(mid_dim, mid_dim, 3, relu=False)
self.bn3 = nn.BatchNorm2d(mid_dim)
self.conv3 = ConvBlock(mid_dim, out_dim, 1, relu=False)
self.skip_layer = ConvBlock(inp_dim, out_dim, 1, relu=False)
if inp_dim == out_dim:
self.need_skip = False
else:
self.need_skip = True
def forward(self, x):
if self.need_skip:
residual = self.skip_layer(x)
else:
residual = x
out = self.bn1(x)
out = self.relu(out)
out = self.conv1(out)
out = self.bn2(out)
out = self.relu(out)
out = self.conv2(out)
out = self.bn3(out)
out = self.relu(out)
out = self.conv3(out)
out += residual
return out
class Hourglass(nn.Module):
def __init__(self, n, f, increase=0, up_mode='nearest',
add_coord=False, first_one=False, x_dim=64, y_dim=64):
super(Hourglass, self).__init__()
nf = f + increase
Block = ResBlock
if add_coord:
self.coordconv = CoordConvTh(x_dim=x_dim, y_dim=y_dim,
with_r=True, with_boundary=True,
relu=False, bn=False,
in_channels=f, out_channels=f,
first_one=first_one,
kernel_size=1,
stride=1, padding=0)
else:
self.coordconv = None
self.up1 = Block(f, f)
# Lower branch
self.pool1 = nn.MaxPool2d(kernel_size=2, stride=2)
self.low1 = Block(f, nf)
self.n = n
# Recursive hourglass
if self.n > 1:
self.low2 = Hourglass(n=n - 1, f=nf, increase=increase, up_mode=up_mode, add_coord=False)
else:
self.low2 = Block(nf, nf)
self.low3 = Block(nf, f)
self.up2 = nn.Upsample(scale_factor=2, mode=up_mode)
def forward(self, x, heatmap=None):
if self.coordconv is not None:
x = self.coordconv(x, heatmap)
up1 = self.up1(x)
pool1 = self.pool1(x)
low1 = self.low1(pool1)
low2 = self.low2(low1)
low3 = self.low3(low2)
up2 = self.up2(low3)
return up1 + up2
class E2HTransform(nn.Module):
def __init__(self, edge_info, num_points, num_edges):
super().__init__()
e2h_matrix = np.zeros([num_points, num_edges])
for edge_id, isclosed_indices in enumerate(edge_info):
is_closed, indices = isclosed_indices
for point_id in indices:
e2h_matrix[point_id, edge_id] = 1
e2h_matrix = torch.from_numpy(e2h_matrix).float()
# pn x en x 1 x 1.
self.register_buffer('weight', e2h_matrix.view(
e2h_matrix.size(0), e2h_matrix.size(1), 1, 1))
# some keypoints are not coverred by any edges,
# in these cases, we must add a constant bias to their heatmap weights.
bias = ((e2h_matrix @ torch.ones(e2h_matrix.size(1)).to(
e2h_matrix)) < 0.5).to(e2h_matrix)
# pn x 1.
self.register_buffer('bias', bias)
def forward(self, edgemaps):
# input: batch_size x en x hw x hh.
# output: batch_size x pn x hw x hh.
return F.conv2d(edgemaps, weight=self.weight, bias=self.bias)
class StackedHGNetV1(nn.Module):
def __init__(self, config, classes_num, edge_info,
nstack=4, nlevels=4, in_channel=256, increase=0,
add_coord=True, decoder_type='default'):
super(StackedHGNetV1, self).__init__()
self.cfg = config
self.coder_type = decoder_type
self.decoder = get_decoder(decoder_type=decoder_type)
self.nstack = nstack
self.add_coord = add_coord
self.num_heats = classes_num[0]
if self.add_coord:
convBlock = CoordConvTh(x_dim=self.cfg.width, y_dim=self.cfg.height,
with_r=True, with_boundary=False,
relu=True, bn=True,
in_channels=3, out_channels=64,
kernel_size=7,
stride=2, padding=3)
else:
convBlock = ConvBlock(3, 64, 7, 2, bn=True, relu=True)
pool = nn.MaxPool2d(kernel_size=2, stride=2)
Block = ResBlock
self.pre = nn.Sequential(
convBlock,
Block(64, 128),
pool,
Block(128, 128),
Block(128, in_channel)
)
self.hgs = nn.ModuleList(
[Hourglass(n=nlevels, f=in_channel, increase=increase, add_coord=self.add_coord, first_one=(_ == 0),
x_dim=int(self.cfg.width / self.nstack), y_dim=int(self.cfg.height / self.nstack))
for _ in range(nstack)])
self.features = nn.ModuleList([
nn.Sequential(
Block(in_channel, in_channel),
ConvBlock(in_channel, in_channel, 1, bn=True, relu=True)
) for _ in range(nstack)])
self.out_heatmaps = nn.ModuleList(
[ConvBlock(in_channel, self.num_heats, 1, relu=False, bn=False)
for _ in range(nstack)])
if self.cfg.use_AAM:
self.num_edges = classes_num[1]
self.num_points = classes_num[2]
self.e2h_transform = E2HTransform(edge_info, self.num_points, self.num_edges)
self.out_edgemaps = nn.ModuleList(
[ConvBlock(in_channel, self.num_edges, 1, relu=False, bn=False)
for _ in range(nstack)])
self.out_pointmaps = nn.ModuleList(
[ConvBlock(in_channel, self.num_points, 1, relu=False, bn=False)
for _ in range(nstack)])
self.merge_edgemaps = nn.ModuleList(
[ConvBlock(self.num_edges, in_channel, 1, relu=False, bn=False)
for _ in range(nstack - 1)])
self.merge_pointmaps = nn.ModuleList(
[ConvBlock(self.num_points, in_channel, 1, relu=False, bn=False)
for _ in range(nstack - 1)])
self.edgemap_act = Activation("sigmoid", self.num_edges)
self.pointmap_act = Activation("sigmoid", self.num_points)
self.merge_features = nn.ModuleList(
[ConvBlock(in_channel, in_channel, 1, relu=False, bn=False)
for _ in range(nstack - 1)])
self.merge_heatmaps = nn.ModuleList(
[ConvBlock(self.num_heats, in_channel, 1, relu=False, bn=False)
for _ in range(nstack - 1)])
self.nstack = nstack
self.heatmap_act = Activation("in+relu", self.num_heats)
self.inference = False
def set_inference(self, inference):
self.inference = inference
def forward(self, x):
x = self.pre(x)
y, fusionmaps = [], []
heatmaps = None
for i in range(self.nstack):
hg = self.hgs[i](x, heatmap=heatmaps)
feature = self.features[i](hg)
heatmaps0 = self.out_heatmaps[i](feature)
heatmaps = self.heatmap_act(heatmaps0)
if self.cfg.use_AAM:
pointmaps0 = self.out_pointmaps[i](feature)
pointmaps = self.pointmap_act(pointmaps0)
edgemaps0 = self.out_edgemaps[i](feature)
edgemaps = self.edgemap_act(edgemaps0)
mask = self.e2h_transform(edgemaps) * pointmaps
fusion_heatmaps = mask * heatmaps
else:
fusion_heatmaps = heatmaps
landmarks = self.decoder.get_coords_from_heatmap(fusion_heatmaps)
if i < self.nstack - 1:
x = x + self.merge_features[i](feature) + \
self.merge_heatmaps[i](heatmaps)
if self.cfg.use_AAM:
x += self.merge_pointmaps[i](pointmaps)
x += self.merge_edgemaps[i](edgemaps)
y.append(landmarks)
if self.cfg.use_AAM:
y.append(pointmaps)
y.append(edgemaps)
fusionmaps.append(fusion_heatmaps)
return y, fusionmaps, landmarks
import numpy as np
import torch
import torch.nn as nn
import torch.nn.functional as F
from ..dataset import get_decoder
from .core.coord_conv import CoordConvTh
class Activation(nn.Module):
def __init__(self, kind: str = 'relu', channel=None):
super().__init__()
self.kind = kind
if '+' in kind:
norm_str, act_str = kind.split('+')
else:
norm_str, act_str = 'none', kind
self.norm_fn = {
'in':
F.instance_norm,
'bn':
nn.BatchNorm2d(channel),
'bn_noaffine':
nn.BatchNorm2d(channel, affine=False, track_running_stats=True),
'none':
None
}[norm_str]
self.act_fn = {
'relu': F.relu,
'softplus': nn.Softplus(),
'exp': torch.exp,
'sigmoid': torch.sigmoid,
'tanh': torch.tanh,
'none': None
}[act_str]
self.channel = channel
def forward(self, x):
if self.norm_fn is not None:
x = self.norm_fn(x)
if self.act_fn is not None:
x = self.act_fn(x)
return x
def extra_repr(self):
return f'kind={self.kind}, channel={self.channel}'
class ConvBlock(nn.Module):
def __init__(self,
inp_dim,
out_dim,
kernel_size=3,
stride=1,
bn=False,
relu=True,
groups=1):
super(ConvBlock, self).__init__()
self.inp_dim = inp_dim
self.conv = nn.Conv2d(
inp_dim,
out_dim,
kernel_size,
stride,
padding=(kernel_size - 1) // 2,
groups=groups,
bias=True)
self.relu = None
self.bn = None
if relu:
self.relu = nn.ReLU()
if bn:
self.bn = nn.BatchNorm2d(out_dim)
def forward(self, x):
x = self.conv(x)
if self.bn is not None:
x = self.bn(x)
if self.relu is not None:
x = self.relu(x)
return x
class ResBlock(nn.Module):
def __init__(self, inp_dim, out_dim, mid_dim=None):
super(ResBlock, self).__init__()
if mid_dim is None:
mid_dim = out_dim // 2
self.relu = nn.ReLU()
self.bn1 = nn.BatchNorm2d(inp_dim)
self.conv1 = ConvBlock(inp_dim, mid_dim, 1, relu=False)
self.bn2 = nn.BatchNorm2d(mid_dim)
self.conv2 = ConvBlock(mid_dim, mid_dim, 3, relu=False)
self.bn3 = nn.BatchNorm2d(mid_dim)
self.conv3 = ConvBlock(mid_dim, out_dim, 1, relu=False)
self.skip_layer = ConvBlock(inp_dim, out_dim, 1, relu=False)
if inp_dim == out_dim:
self.need_skip = False
else:
self.need_skip = True
def forward(self, x):
if self.need_skip:
residual = self.skip_layer(x)
else:
residual = x
out = self.bn1(x)
out = self.relu(out)
out = self.conv1(out)
out = self.bn2(out)
out = self.relu(out)
out = self.conv2(out)
out = self.bn3(out)
out = self.relu(out)
out = self.conv3(out)
out += residual
return out
class Hourglass(nn.Module):
def __init__(self,
n,
f,
increase=0,
up_mode='nearest',
add_coord=False,
first_one=False,
x_dim=64,
y_dim=64):
super(Hourglass, self).__init__()
nf = f + increase
Block = ResBlock
if add_coord:
self.coordconv = CoordConvTh(
x_dim=x_dim,
y_dim=y_dim,
with_r=True,
with_boundary=True,
relu=False,
bn=False,
in_channels=f,
out_channels=f,
first_one=first_one,
kernel_size=1,
stride=1,
padding=0)
else:
self.coordconv = None
self.up1 = Block(f, f)
# Lower branch
self.pool1 = nn.MaxPool2d(kernel_size=2, stride=2)
self.low1 = Block(f, nf)
self.n = n
# Recursive hourglass
if self.n > 1:
self.low2 = Hourglass(
n=n - 1,
f=nf,
increase=increase,
up_mode=up_mode,
add_coord=False)
else:
self.low2 = Block(nf, nf)
self.low3 = Block(nf, f)
self.up2 = nn.Upsample(scale_factor=2, mode=up_mode)
def forward(self, x, heatmap=None):
if self.coordconv is not None:
x = self.coordconv(x, heatmap)
up1 = self.up1(x)
pool1 = self.pool1(x)
low1 = self.low1(pool1)
low2 = self.low2(low1)
low3 = self.low3(low2)
up2 = self.up2(low3)
return up1 + up2
class E2HTransform(nn.Module):
def __init__(self, edge_info, num_points, num_edges):
super().__init__()
e2h_matrix = np.zeros([num_points, num_edges])
for edge_id, isclosed_indices in enumerate(edge_info):
is_closed, indices = isclosed_indices
for point_id in indices:
e2h_matrix[point_id, edge_id] = 1
e2h_matrix = torch.from_numpy(e2h_matrix).float()
# pn x en x 1 x 1.
self.register_buffer(
'weight',
e2h_matrix.view(e2h_matrix.size(0), e2h_matrix.size(1), 1, 1))
# some keypoints are not coverred by any edges,
# in these cases, we must add a constant bias to their heatmap weights.
bias = ((e2h_matrix @ torch.ones(e2h_matrix.size(1)).to(e2h_matrix))
< # noqa
0.5).to(e2h_matrix) # noqa
# pn x 1.
self.register_buffer('bias', bias)
def forward(self, edgemaps):
# input: batch_size x en x hw x hh.
# output: batch_size x pn x hw x hh.
return F.conv2d(edgemaps, weight=self.weight, bias=self.bias)
class StackedHGNetV1(nn.Module):
def __init__(self,
config,
classes_num,
edge_info,
nstack=4,
nlevels=4,
in_channel=256,
increase=0,
add_coord=True,
decoder_type='default'):
super(StackedHGNetV1, self).__init__()
self.cfg = config
self.coder_type = decoder_type
self.decoder = get_decoder(decoder_type=decoder_type)
self.nstack = nstack
self.add_coord = add_coord
self.num_heats = classes_num[0]
if self.add_coord:
convBlock = CoordConvTh(
x_dim=self.cfg.width,
y_dim=self.cfg.height,
with_r=True,
with_boundary=False,
relu=True,
bn=True,
in_channels=3,
out_channels=64,
kernel_size=7,
stride=2,
padding=3)
else:
convBlock = ConvBlock(3, 64, 7, 2, bn=True, relu=True)
pool = nn.MaxPool2d(kernel_size=2, stride=2)
Block = ResBlock
self.pre = nn.Sequential(convBlock, Block(64, 128), pool,
Block(128, 128), Block(128, in_channel))
self.hgs = nn.ModuleList([
Hourglass(
n=nlevels,
f=in_channel,
increase=increase,
add_coord=self.add_coord,
first_one=(_ == 0),
x_dim=int(self.cfg.width / self.nstack),
y_dim=int(self.cfg.height / self.nstack))
for _ in range(nstack)
])
self.features = nn.ModuleList([
nn.Sequential(
Block(in_channel, in_channel),
ConvBlock(in_channel, in_channel, 1, bn=True, relu=True))
for _ in range(nstack)
])
self.out_heatmaps = nn.ModuleList([
ConvBlock(in_channel, self.num_heats, 1, relu=False, bn=False)
for _ in range(nstack)
])
if self.cfg.use_AAM:
self.num_edges = classes_num[1]
self.num_points = classes_num[2]
self.e2h_transform = E2HTransform(edge_info, self.num_points,
self.num_edges)
self.out_edgemaps = nn.ModuleList([
ConvBlock(in_channel, self.num_edges, 1, relu=False, bn=False)
for _ in range(nstack)
])
self.out_pointmaps = nn.ModuleList([
ConvBlock(
in_channel, self.num_points, 1, relu=False, bn=False)
for _ in range(nstack)
])
self.merge_edgemaps = nn.ModuleList([
ConvBlock(self.num_edges, in_channel, 1, relu=False, bn=False)
for _ in range(nstack - 1)
])
self.merge_pointmaps = nn.ModuleList([
ConvBlock(
self.num_points, in_channel, 1, relu=False, bn=False)
for _ in range(nstack - 1)
])
self.edgemap_act = Activation('sigmoid', self.num_edges)
self.pointmap_act = Activation('sigmoid', self.num_points)
self.merge_features = nn.ModuleList([
ConvBlock(in_channel, in_channel, 1, relu=False, bn=False)
for _ in range(nstack - 1)
])
self.merge_heatmaps = nn.ModuleList([
ConvBlock(self.num_heats, in_channel, 1, relu=False, bn=False)
for _ in range(nstack - 1)
])
self.nstack = nstack
self.heatmap_act = Activation('in+relu', self.num_heats)
self.inference = False
def set_inference(self, inference):
self.inference = inference
def forward(self, x):
x = self.pre(x)
y, fusionmaps = [], []
heatmaps = None
for i in range(self.nstack):
hg = self.hgs[i](x, heatmap=heatmaps)
feature = self.features[i](hg)
heatmaps0 = self.out_heatmaps[i](feature)
heatmaps = self.heatmap_act(heatmaps0)
if self.cfg.use_AAM:
pointmaps0 = self.out_pointmaps[i](feature)
pointmaps = self.pointmap_act(pointmaps0)
edgemaps0 = self.out_edgemaps[i](feature)
edgemaps = self.edgemap_act(edgemaps0)
mask = self.e2h_transform(edgemaps) * pointmaps
fusion_heatmaps = mask * heatmaps
else:
fusion_heatmaps = heatmaps
landmarks = self.decoder.get_coords_from_heatmap(fusion_heatmaps)
if i < self.nstack - 1:
x = x + self.merge_features[i](feature) + \
self.merge_heatmaps[i](heatmaps)
if self.cfg.use_AAM:
x += self.merge_pointmaps[i](pointmaps)
x += self.merge_edgemaps[i](edgemaps)
y.append(landmarks)
if self.cfg.use_AAM:
y.append(pointmaps)
y.append(edgemaps)
fusionmaps.append(fusion_heatmaps)
return y, fusionmaps, landmarks

View File

@@ -1,10 +1,5 @@
from .encoder import get_encoder
from .decoder import get_decoder
from .alignmentDataset import AlignmentDataset
__all__ = [
"Augmentation",
"AlignmentDataset",
"get_encoder",
"get_decoder"
]
from .alignmentDataset import AlignmentDataset
from .decoder import get_decoder
from .encoder import get_encoder
__all__ = ['Augmentation', 'AlignmentDataset', 'get_encoder', 'get_decoder']

View File

@@ -1,314 +1,360 @@
import os
import sys
import cv2
import math
import copy
import hashlib
import imageio
import numpy as np
import pandas as pd
from scipy import interpolate
from PIL import Image, ImageEnhance, ImageFile
import torch
import torch.nn.functional as F
from torch.utils.data import Dataset
ImageFile.LOAD_TRUNCATED_IMAGES = True
from .encoder import get_encoder
class AlignmentDataset(Dataset):
def __init__(self, tsv_flie, image_dir="", transform=None,
width=256, height=256, channels=3,
means=(127.5, 127.5, 127.5), scale=1 / 127.5,
classes_num=None, crop_op=True, aug_prob=0.0, edge_info=None, flip_mapping=None, is_train=True,
encoder_type='default',
):
super(AlignmentDataset, self).__init__()
self.use_AAM = True
self.encoder_type = encoder_type
self.encoder = get_encoder(height, width, encoder_type=encoder_type)
self.items = pd.read_csv(tsv_flie, sep="\t")
self.image_dir = image_dir
self.landmark_num = classes_num[0]
self.transform = transform
self.image_width = width
self.image_height = height
self.channels = channels
assert self.image_width == self.image_height
self.means = means
self.scale = scale
self.aug_prob = aug_prob
self.edge_info = edge_info
self.is_train = is_train
std_lmk_5pts = np.array([
196.0, 226.0,
316.0, 226.0,
256.0, 286.0,
220.0, 360.4,
292.0, 360.4], np.float32) / 256.0 - 1.0
std_lmk_5pts = np.reshape(std_lmk_5pts, (5, 2)) # [-1 1]
target_face_scale = 1.0 if crop_op else 1.25
self.augmentation = Augmentation(
is_train=self.is_train,
aug_prob=self.aug_prob,
image_size=self.image_width,
crop_op=crop_op,
std_lmk_5pts=std_lmk_5pts,
target_face_scale=target_face_scale,
flip_rate=0.5,
flip_mapping=flip_mapping,
random_shift_sigma=0.05,
random_rot_sigma=math.pi / 180 * 18,
random_scale_sigma=0.1,
random_gray_rate=0.2,
random_occ_rate=0.4,
random_blur_rate=0.3,
random_gamma_rate=0.2,
random_nose_fusion_rate=0.2)
def _circle(self, img, pt, sigma=1.0, label_type='Gaussian'):
# Check that any part of the gaussian is in-bounds
tmp_size = sigma * 3
ul = [int(pt[0] - tmp_size), int(pt[1] - tmp_size)]
br = [int(pt[0] + tmp_size + 1), int(pt[1] + tmp_size + 1)]
if (ul[0] > img.shape[1] - 1 or ul[1] > img.shape[0] - 1 or
br[0] - 1 < 0 or br[1] - 1 < 0):
# If not, just return the image as is
return img
# Generate gaussian
size = 2 * tmp_size + 1
x = np.arange(0, size, 1, np.float32)
y = x[:, np.newaxis]
x0 = y0 = size // 2
# The gaussian is not normalized, we want the center value to equal 1
if label_type == 'Gaussian':
g = np.exp(- ((x - x0) ** 2 + (y - y0) ** 2) / (2 * sigma ** 2))
else:
g = sigma / (((x - x0) ** 2 + (y - y0) ** 2 + sigma ** 2) ** 1.5)
# Usable gaussian range
g_x = max(0, -ul[0]), min(br[0], img.shape[1]) - ul[0]
g_y = max(0, -ul[1]), min(br[1], img.shape[0]) - ul[1]
# Image range
img_x = max(0, ul[0]), min(br[0], img.shape[1])
img_y = max(0, ul[1]), min(br[1], img.shape[0])
img[img_y[0]:img_y[1], img_x[0]:img_x[1]] = 255 * g[g_y[0]:g_y[1], g_x[0]:g_x[1]]
return img
def _polylines(self, img, lmks, is_closed, color=255, thickness=1, draw_mode=cv2.LINE_AA,
interpolate_mode=cv2.INTER_AREA, scale=4):
h, w = img.shape
img_scale = cv2.resize(img, (w * scale, h * scale), interpolation=interpolate_mode)
lmks_scale = (lmks * scale + 0.5).astype(np.int32)
cv2.polylines(img_scale, [lmks_scale], is_closed, color, thickness * scale, draw_mode)
img = cv2.resize(img_scale, (w, h), interpolation=interpolate_mode)
return img
def _generate_edgemap(self, points, scale=0.25, thickness=1):
h, w = self.image_height, self.image_width
edgemaps = []
for is_closed, indices in self.edge_info:
edgemap = np.zeros([h, w], dtype=np.float32)
# align_corners: False.
part = copy.deepcopy(points[np.array(indices)])
part = self._fit_curve(part, is_closed)
part[:, 0] = np.clip(part[:, 0], 0, w - 1)
part[:, 1] = np.clip(part[:, 1], 0, h - 1)
edgemap = self._polylines(edgemap, part, is_closed, 255, thickness)
edgemaps.append(edgemap)
edgemaps = np.stack(edgemaps, axis=0) / 255.0
edgemaps = torch.from_numpy(edgemaps).float().unsqueeze(0)
edgemaps = F.interpolate(edgemaps, size=(int(w * scale), int(h * scale)), mode='bilinear',
align_corners=False).squeeze()
return edgemaps
def _fit_curve(self, lmks, is_closed=False, density=5):
try:
x = lmks[:, 0].copy()
y = lmks[:, 1].copy()
if is_closed:
x = np.append(x, x[0])
y = np.append(y, y[0])
tck, u = interpolate.splprep([x, y], s=0, per=is_closed, k=3)
# bins = (x.shape[0] - 1) * density + 1
# lmk_x, lmk_y = interpolate.splev(np.linspace(0, 1, bins), f)
intervals = np.array([])
for i in range(len(u) - 1):
intervals = np.concatenate((intervals, np.linspace(u[i], u[i + 1], density, endpoint=False)))
if not is_closed:
intervals = np.concatenate((intervals, [u[-1]]))
lmk_x, lmk_y = interpolate.splev(intervals, tck, der=0)
# der_x, der_y = interpolate.splev(intervals, tck, der=1)
curve_lmks = np.stack([lmk_x, lmk_y], axis=-1)
# curve_ders = np.stack([der_x, der_y], axis=-1)
# origin_indices = np.arange(0, curve_lmks.shape[0], density)
return curve_lmks
except:
return lmks
def _image_id(self, image_path):
if not os.path.exists(image_path):
image_path = os.path.join(self.image_dir, image_path)
return hashlib.md5(open(image_path, "rb").read()).hexdigest()
def _load_image(self, image_path):
if not os.path.exists(image_path):
image_path = os.path.join(self.image_dir, image_path)
try:
# img = cv2.imdecode(np.fromfile(image_path, dtype=np.uint8), cv2.IMREAD_COLOR)#HWC, BGR, [0-255]
img = cv2.imread(image_path, cv2.IMREAD_COLOR) # HWC, BGR, [0-255]
assert img is not None and len(img.shape) == 3 and img.shape[2] == 3
except:
try:
img = imageio.imread(image_path) # HWC, RGB, [0-255]
img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR) # HWC, BGR, [0-255]
assert img is not None and len(img.shape) == 3 and img.shape[2] == 3
except:
try:
gifImg = imageio.mimread(image_path) # BHWC, RGB, [0-255]
img = gifImg[0] # HWC, RGB, [0-255]
img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR) # HWC, BGR, [0-255]
assert img is not None and len(img.shape) == 3 and img.shape[2] == 3
except:
img = None
return img
def _compose_rotate_and_scale(self, angle, scale, shift_xy, from_center, to_center):
cosv = math.cos(angle)
sinv = math.sin(angle)
fx, fy = from_center
tx, ty = to_center
acos = scale * cosv
asin = scale * sinv
a0 = acos
a1 = -asin
a2 = tx - acos * fx + asin * fy + shift_xy[0]
b0 = asin
b1 = acos
b2 = ty - asin * fx - acos * fy + shift_xy[1]
rot_scale_m = np.array([
[a0, a1, a2],
[b0, b1, b2],
[0.0, 0.0, 1.0]
], np.float32)
return rot_scale_m
def _transformPoints2D(self, points, matrix):
"""
points (nx2), matrix (3x3) -> points (nx2)
"""
dtype = points.dtype
# nx3
points = np.concatenate([points, np.ones_like(points[:, [0]])], axis=1)
points = points @ np.transpose(matrix) # nx3
points = points[:, :2] / points[:, [2, 2]]
return points.astype(dtype)
def _transformPerspective(self, image, matrix, target_shape):
"""
image, matrix3x3 -> transformed_image
"""
return cv2.warpPerspective(
image, matrix,
dsize=(target_shape[1], target_shape[0]),
flags=cv2.INTER_LINEAR, borderValue=0)
def _norm_points(self, points, h, w, align_corners=False):
if align_corners:
# [0, SIZE-1] -> [-1, +1]
des_points = points / torch.tensor([w - 1, h - 1]).to(points).view(1, 2) * 2 - 1
else:
# [-0.5, SIZE-0.5] -> [-1, +1]
des_points = (points * 2 + 1) / torch.tensor([w, h]).to(points).view(1, 2) - 1
des_points = torch.clamp(des_points, -1, 1)
return des_points
def _denorm_points(self, points, h, w, align_corners=False):
if align_corners:
# [-1, +1] -> [0, SIZE-1]
des_points = (points + 1) / 2 * torch.tensor([w - 1, h - 1]).to(points).view(1, 1, 2)
else:
# [-1, +1] -> [-0.5, SIZE-0.5]
des_points = ((points + 1) * torch.tensor([w, h]).to(points).view(1, 1, 2) - 1) / 2
return des_points
def __len__(self):
return len(self.items)
def __getitem__(self, index):
sample = dict()
image_path = self.items.iloc[index, 0]
landmarks_5pts = self.items.iloc[index, 1]
landmarks_5pts = np.array(list(map(float, landmarks_5pts.split(","))), dtype=np.float32).reshape(5, 2)
landmarks_target = self.items.iloc[index, 2]
landmarks_target = np.array(list(map(float, landmarks_target.split(","))), dtype=np.float32).reshape(
self.landmark_num, 2)
scale = float(self.items.iloc[index, 3])
center_w, center_h = float(self.items.iloc[index, 4]), float(self.items.iloc[index, 5])
if len(self.items.iloc[index]) > 6:
tags = np.array(list(map(lambda x: int(float(x)), self.items.iloc[index, 6].split(","))))
else:
tags = np.array([])
# image & keypoints alignment
image_path = image_path.replace('\\', '/')
# wflw testset
image_path = image_path.replace(
'//msr-facestore/Workspace/MSRA_EP_Allergan/users/yanghuan/training_data/wflw/rawImages/', '')
# trainset
image_path = image_path.replace('./rawImages/', '')
image_path = os.path.join(self.image_dir, image_path)
# image path
sample["image_path"] = image_path
img = self._load_image(image_path) # HWC, BGR, [0, 255]
assert img is not None
# augmentation
# landmarks_target = [-0.5, edge-0.5]
img, landmarks_target, matrix = \
self.augmentation.process(img, landmarks_target, landmarks_5pts, scale, center_w, center_h)
landmarks = self._norm_points(torch.from_numpy(landmarks_target), self.image_height, self.image_width)
sample["label"] = [landmarks, ]
if self.use_AAM:
pointmap = self.encoder.generate_heatmap(landmarks_target)
edgemap = self._generate_edgemap(landmarks_target)
sample["label"] += [pointmap, edgemap]
sample['matrix'] = matrix
# image normalization
img = img.transpose(2, 0, 1).astype(np.float32) # CHW, BGR, [0, 255]
img[0, :, :] = (img[0, :, :] - self.means[0]) * self.scale
img[1, :, :] = (img[1, :, :] - self.means[1]) * self.scale
img[2, :, :] = (img[2, :, :] - self.means[2]) * self.scale
sample["data"] = torch.from_numpy(img) # CHW, BGR, [-1, 1]
sample["tags"] = tags
return sample
import copy
import hashlib
import math
import os
import sys
import cv2
import imageio
import numpy as np
import pandas as pd
import torch
import torch.nn.functional as F
from PIL import Image, ImageEnhance, ImageFile
from scipy import interpolate
from torch.utils.data import Dataset
from .encoder import get_encoder
ImageFile.LOAD_TRUNCATED_IMAGES = True
class AlignmentDataset(Dataset):
def __init__(
self,
tsv_flie,
image_dir='',
transform=None,
width=256,
height=256,
channels=3,
means=(127.5, 127.5, 127.5),
scale=1 / 127.5,
classes_num=None,
crop_op=True,
aug_prob=0.0,
edge_info=None,
flip_mapping=None,
is_train=True,
encoder_type='default',
):
super(AlignmentDataset, self).__init__()
self.use_AAM = True
self.encoder_type = encoder_type
self.encoder = get_encoder(height, width, encoder_type=encoder_type)
self.items = pd.read_csv(tsv_flie, sep='\t')
self.image_dir = image_dir
self.landmark_num = classes_num[0]
self.transform = transform
self.image_width = width
self.image_height = height
self.channels = channels
assert self.image_width == self.image_height
self.means = means
self.scale = scale
self.aug_prob = aug_prob
self.edge_info = edge_info
self.is_train = is_train
std_lmk_5pts = np.array([
196.0, 226.0, 316.0, 226.0, 256.0, 286.0, 220.0, 360.4, 292.0,
360.4
], np.float32) / 256.0 - 1.0
std_lmk_5pts = np.reshape(std_lmk_5pts, (5, 2)) # [-1 1]
target_face_scale = 1.0 if crop_op else 1.25
self.augmentation = Augmentation(
is_train=self.is_train,
aug_prob=self.aug_prob,
image_size=self.image_width,
crop_op=crop_op,
std_lmk_5pts=std_lmk_5pts,
target_face_scale=target_face_scale,
flip_rate=0.5,
flip_mapping=flip_mapping,
random_shift_sigma=0.05,
random_rot_sigma=math.pi / 180 * 18,
random_scale_sigma=0.1,
random_gray_rate=0.2,
random_occ_rate=0.4,
random_blur_rate=0.3,
random_gamma_rate=0.2,
random_nose_fusion_rate=0.2)
def _circle(self, img, pt, sigma=1.0, label_type='Gaussian'):
# Check that any part of the gaussian is in-bounds
tmp_size = sigma * 3
ul = [int(pt[0] - tmp_size), int(pt[1] - tmp_size)]
br = [int(pt[0] + tmp_size + 1), int(pt[1] + tmp_size + 1)]
if (ul[0] > img.shape[1] - 1 or ul[1] > img.shape[0] - 1
or br[0] - 1 < 0 or br[1] - 1 < 0):
# If not, just return the image as is
return img
# Generate gaussian
size = 2 * tmp_size + 1
x = np.arange(0, size, 1, np.float32)
y = x[:, np.newaxis]
x0 = y0 = size // 2
# The gaussian is not normalized, we want the center value to equal 1
if label_type == 'Gaussian':
g = np.exp(-((x - x0)**2 + (y - y0)**2) / (2 * sigma**2))
else:
g = sigma / (((x - x0)**2 + (y - y0)**2 + sigma**2)**1.5)
# Usable gaussian range
g_x = max(0, -ul[0]), min(br[0], img.shape[1]) - ul[0]
g_y = max(0, -ul[1]), min(br[1], img.shape[0]) - ul[1]
# Image range
img_x = max(0, ul[0]), min(br[0], img.shape[1])
img_y = max(0, ul[1]), min(br[1], img.shape[0])
img[img_y[0]:img_y[1],
img_x[0]:img_x[1]] = 255 * g[g_y[0]:g_y[1], g_x[0]:g_x[1]]
return img
def _polylines(self,
img,
lmks,
is_closed,
color=255,
thickness=1,
draw_mode=cv2.LINE_AA,
interpolate_mode=cv2.INTER_AREA,
scale=4):
h, w = img.shape
img_scale = cv2.resize(
img, (w * scale, h * scale), interpolation=interpolate_mode)
lmks_scale = (lmks * scale + 0.5).astype(np.int32)
cv2.polylines(img_scale, [lmks_scale], is_closed, color,
thickness * scale, draw_mode)
img = cv2.resize(img_scale, (w, h), interpolation=interpolate_mode)
return img
def _generate_edgemap(self, points, scale=0.25, thickness=1):
h, w = self.image_height, self.image_width
edgemaps = []
for is_closed, indices in self.edge_info:
edgemap = np.zeros([h, w], dtype=np.float32)
# align_corners: False.
part = copy.deepcopy(points[np.array(indices)])
part = self._fit_curve(part, is_closed)
part[:, 0] = np.clip(part[:, 0], 0, w - 1)
part[:, 1] = np.clip(part[:, 1], 0, h - 1)
edgemap = self._polylines(edgemap, part, is_closed, 255, thickness)
edgemaps.append(edgemap)
edgemaps = np.stack(edgemaps, axis=0) / 255.0
edgemaps = torch.from_numpy(edgemaps).float().unsqueeze(0)
edgemaps = F.interpolate(
edgemaps,
size=(int(w * scale), int(h * scale)),
mode='bilinear',
align_corners=False).squeeze()
return edgemaps
def _fit_curve(self, lmks, is_closed=False, density=5):
try:
x = lmks[:, 0].copy()
y = lmks[:, 1].copy()
if is_closed:
x = np.append(x, x[0])
y = np.append(y, y[0])
tck, u = interpolate.splprep([x, y], s=0, per=is_closed, k=3)
# bins = (x.shape[0] - 1) * density + 1
# lmk_x, lmk_y = interpolate.splev(np.linspace(0, 1, bins), f)
intervals = np.array([])
for i in range(len(u) - 1):
intervals = np.concatenate(
(intervals,
np.linspace(u[i], u[i + 1], density, endpoint=False)))
if not is_closed:
intervals = np.concatenate((intervals, [u[-1]]))
lmk_x, lmk_y = interpolate.splev(intervals, tck, der=0)
# der_x, der_y = interpolate.splev(intervals, tck, der=1)
curve_lmks = np.stack([lmk_x, lmk_y], axis=-1)
# curve_ders = np.stack([der_x, der_y], axis=-1)
# origin_indices = np.arange(0, curve_lmks.shape[0], density)
return curve_lmks
except Exception:
return lmks
def _image_id(self, image_path):
if not os.path.exists(image_path):
image_path = os.path.join(self.image_dir, image_path)
return hashlib.md5(open(image_path, 'rb').read()).hexdigest()
def _load_image(self, image_path):
if not os.path.exists(image_path):
image_path = os.path.join(self.image_dir, image_path)
try:
# img = cv2.imdecode(np.fromfile(image_path, dtype=np.uint8), cv2.IMREAD_COLOR)#HWC, BGR, [0-255]
img = cv2.imread(image_path, cv2.IMREAD_COLOR) # HWC, BGR, [0-255]
assert img is not None and len(
img.shape) == 3 and img.shape[2] == 3
except Exception:
try:
img = imageio.imread(image_path) # HWC, RGB, [0-255]
img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR) # HWC, BGR, [0-255]
assert img is not None and len(
img.shape) == 3 and img.shape[2] == 3
except Exception:
try:
gifImg = imageio.mimread(image_path) # BHWC, RGB, [0-255]
img = gifImg[0] # HWC, RGB, [0-255]
img = cv2.cvtColor(img,
cv2.COLOR_RGB2BGR) # HWC, BGR, [0-255]
assert img is not None and len(
img.shape) == 3 and img.shape[2] == 3
except Exception:
img = None
return img
def _compose_rotate_and_scale(self, angle, scale, shift_xy, from_center,
to_center):
cosv = math.cos(angle)
sinv = math.sin(angle)
fx, fy = from_center
tx, ty = to_center
acos = scale * cosv
asin = scale * sinv
a0 = acos
a1 = -asin
a2 = tx - acos * fx + asin * fy + shift_xy[0]
b0 = asin
b1 = acos
b2 = ty - asin * fx - acos * fy + shift_xy[1]
rot_scale_m = np.array([[a0, a1, a2], [b0, b1, b2], [0.0, 0.0, 1.0]],
np.float32)
return rot_scale_m
def _transformPoints2D(self, points, matrix):
"""
points (nx2), matrix (3x3) -> points (nx2)
"""
dtype = points.dtype
# nx3
points = np.concatenate([points, np.ones_like(points[:, [0]])], axis=1)
points = points @ np.transpose(matrix) # nx3
points = points[:, :2] / points[:, [2, 2]]
return points.astype(dtype)
def _transformPerspective(self, image, matrix, target_shape):
"""
image, matrix3x3 -> transformed_image
"""
return cv2.warpPerspective(
image,
matrix,
dsize=(target_shape[1], target_shape[0]),
flags=cv2.INTER_LINEAR,
borderValue=0)
def _norm_points(self, points, h, w, align_corners=False):
if align_corners:
# [0, SIZE-1] -> [-1, +1]
des_points = points / torch.tensor([w - 1, h - 1]).to(points).view(
1, 2) * 2 - 1
else:
# [-0.5, SIZE-0.5] -> [-1, +1]
des_points = (points * 2 + 1) / torch.tensor(
[w, h]).to(points).view(1, 2) - 1
des_points = torch.clamp(des_points, -1, 1)
return des_points
def _denorm_points(self, points, h, w, align_corners=False):
if align_corners:
# [-1, +1] -> [0, SIZE-1]
des_points = (points + 1) / 2 * torch.tensor(
[w - 1, h - 1]).to(points).view(1, 1, 2)
else:
# [-1, +1] -> [-0.5, SIZE-0.5]
des_points = (
(points + 1) * torch.tensor([w, h]).to(points).view(1, 1, 2)
- 1) / 2
return des_points
def __len__(self):
return len(self.items)
def __getitem__(self, index):
sample = dict()
image_path = self.items.iloc[index, 0]
landmarks_5pts = self.items.iloc[index, 1]
landmarks_5pts = np.array(
list(map(float, landmarks_5pts.split(','))),
dtype=np.float32).reshape(5, 2)
landmarks_target = self.items.iloc[index, 2]
landmarks_target = np.array(
list(map(float, landmarks_target.split(','))),
dtype=np.float32).reshape(self.landmark_num, 2)
scale = float(self.items.iloc[index, 3])
center_w, center_h = float(self.items.iloc[index, 4]), float(
self.items.iloc[index, 5])
if len(self.items.iloc[index]) > 6:
tags = np.array(
list(
map(lambda x: int(float(x)),
self.items.iloc[index, 6].split(','))))
else:
tags = np.array([])
# image & keypoints alignment
image_path = image_path.replace('\\', '/')
# wflw testset
image_path = image_path.replace(
'//msr-facestore/Workspace/MSRA_EP_Allergan/users/yanghuan/training_data/wflw/rawImages/',
'')
# trainset
image_path = image_path.replace('./rawImages/', '')
image_path = os.path.join(self.image_dir, image_path)
# image path
sample['image_path'] = image_path
img = self._load_image(image_path) # HWC, BGR, [0, 255]
assert img is not None
# augmentation
# landmarks_target = [-0.5, edge-0.5]
img, landmarks_target, matrix = \
self.augmentation.process(img, landmarks_target, landmarks_5pts, scale, center_w, center_h)
landmarks = self._norm_points(
torch.from_numpy(landmarks_target), self.image_height,
self.image_width)
sample['label'] = [
landmarks,
]
if self.use_AAM:
pointmap = self.encoder.generate_heatmap(landmarks_target)
edgemap = self._generate_edgemap(landmarks_target)
sample['label'] += [pointmap, edgemap]
sample['matrix'] = matrix
# image normalization
img = img.transpose(2, 0, 1).astype(np.float32) # CHW, BGR, [0, 255]
img[0, :, :] = (img[0, :, :] - self.means[0]) * self.scale
img[1, :, :] = (img[1, :, :] - self.means[1]) * self.scale
img[2, :, :] = (img[2, :, :] - self.means[2]) * self.scale
sample['data'] = torch.from_numpy(img) # CHW, BGR, [-1, 1]
sample['tags'] = tags
return sample

View File

@@ -1,8 +1,9 @@
from .decoder_default import decoder_default
def get_decoder(decoder_type='default'):
if decoder_type == 'default':
decoder = decoder_default()
else:
raise NotImplementedError
return decoder
from .decoder_default import decoder_default
def get_decoder(decoder_type='default'):
if decoder_type == 'default':
decoder = decoder_default()
else:
raise NotImplementedError
return decoder

View File

@@ -1,38 +1,39 @@
import torch
class decoder_default:
def __init__(self, weight=1, use_weight_map=False):
self.weight = weight
self.use_weight_map = use_weight_map
def _make_grid(self, h, w):
yy, xx = torch.meshgrid(
torch.arange(h).float() / (h - 1) * 2 - 1,
torch.arange(w).float() / (w - 1) * 2 - 1)
return yy, xx
def get_coords_from_heatmap(self, heatmap):
"""
inputs:
- heatmap: batch x npoints x h x w
outputs:
- coords: batch x npoints x 2 (x,y), [-1, +1]
- radius_sq: batch x npoints
"""
batch, npoints, h, w = heatmap.shape
if self.use_weight_map:
heatmap = heatmap * self.weight
yy, xx = self._make_grid(h, w)
yy = yy.view(1, 1, h, w).to(heatmap)
xx = xx.view(1, 1, h, w).to(heatmap)
heatmap_sum = torch.clamp(heatmap.sum([2, 3]), min=1e-6)
yy_coord = (yy * heatmap).sum([2, 3]) / heatmap_sum # batch x npoints
xx_coord = (xx * heatmap).sum([2, 3]) / heatmap_sum # batch x npoints
coords = torch.stack([xx_coord, yy_coord], dim=-1)
return coords
import torch
class decoder_default:
def __init__(self, weight=1, use_weight_map=False):
self.weight = weight
self.use_weight_map = use_weight_map
def _make_grid(self, h, w):
yy, xx = torch.meshgrid(
torch.arange(h).float() / (h - 1) * 2 - 1,
torch.arange(w).float() / (w - 1) * 2 - 1)
return yy, xx
def get_coords_from_heatmap(self, heatmap):
"""
inputs:
- heatmap: batch x npoints x h x w
outputs:
- coords: batch x npoints x 2 (x,y), [-1, +1]
- radius_sq: batch x npoints
"""
batch, npoints, h, w = heatmap.shape
if self.use_weight_map:
heatmap = heatmap * self.weight
yy, xx = self._make_grid(h, w)
yy = yy.view(1, 1, h, w).to(heatmap)
xx = xx.view(1, 1, h, w).to(heatmap)
heatmap_sum = torch.clamp(heatmap.sum([2, 3]), min=1e-6)
yy_coord = (yy * heatmap).sum([2, 3]) / heatmap_sum # batch x npoints
xx_coord = (xx * heatmap).sum([2, 3]) / heatmap_sum # batch x npoints
coords = torch.stack([xx_coord, yy_coord], dim=-1)
return coords

View File

@@ -1,8 +1,13 @@
from .encoder_default import encoder_default
def get_encoder(image_height, image_width, scale=0.25, sigma=1.5, encoder_type='default'):
if encoder_type == 'default':
encoder = encoder_default(image_height, image_width, scale, sigma)
else:
raise NotImplementedError
return encoder
from .encoder_default import encoder_default
def get_encoder(image_height,
image_width,
scale=0.25,
sigma=1.5,
encoder_type='default'):
if encoder_type == 'default':
encoder = encoder_default(image_height, image_width, scale, sigma)
else:
raise NotImplementedError
return encoder

View File

@@ -1,63 +1,68 @@
import copy
import numpy as np
import torch
import torch.nn.functional as F
class encoder_default:
def __init__(self, image_height, image_width, scale=0.25, sigma=1.5):
self.image_height = image_height
self.image_width = image_width
self.scale = scale
self.sigma = sigma
def generate_heatmap(self, points):
# points = (num_pts, 2)
h, w = self.image_height, self.image_width
pointmaps = []
for i in range(len(points)):
pointmap = np.zeros([h, w], dtype=np.float32)
# align_corners: False.
point = copy.deepcopy(points[i])
point[0] = max(0, min(w - 1, point[0]))
point[1] = max(0, min(h - 1, point[1]))
pointmap = self._circle(pointmap, point, sigma=self.sigma)
pointmaps.append(pointmap)
pointmaps = np.stack(pointmaps, axis=0) / 255.0
pointmaps = torch.from_numpy(pointmaps).float().unsqueeze(0)
pointmaps = F.interpolate(pointmaps, size=(int(w * self.scale), int(h * self.scale)), mode='bilinear',
align_corners=False).squeeze()
return pointmaps
def _circle(self, img, pt, sigma=1.0, label_type='Gaussian'):
# Check that any part of the gaussian is in-bounds
tmp_size = sigma * 3
ul = [int(pt[0] - tmp_size), int(pt[1] - tmp_size)]
br = [int(pt[0] + tmp_size + 1), int(pt[1] + tmp_size + 1)]
if (ul[0] > img.shape[1] - 1 or ul[1] > img.shape[0] - 1 or
br[0] - 1 < 0 or br[1] - 1 < 0):
# If not, just return the image as is
return img
# Generate gaussian
size = 2 * tmp_size + 1
x = np.arange(0, size, 1, np.float32)
y = x[:, np.newaxis]
x0 = y0 = size // 2
# The gaussian is not normalized, we want the center value to equal 1
if label_type == 'Gaussian':
g = np.exp(- ((x - x0) ** 2 + (y - y0) ** 2) / (2 * sigma ** 2))
else:
g = sigma / (((x - x0) ** 2 + (y - y0) ** 2 + sigma ** 2) ** 1.5)
# Usable gaussian range
g_x = max(0, -ul[0]), min(br[0], img.shape[1]) - ul[0]
g_y = max(0, -ul[1]), min(br[1], img.shape[0]) - ul[1]
# Image range
img_x = max(0, ul[0]), min(br[0], img.shape[1])
img_y = max(0, ul[1]), min(br[1], img.shape[0])
img[img_y[0]:img_y[1], img_x[0]:img_x[1]] = 255 * g[g_y[0]:g_y[1], g_x[0]:g_x[1]]
return img
import copy
import numpy as np
import torch
import torch.nn.functional as F
class encoder_default:
def __init__(self, image_height, image_width, scale=0.25, sigma=1.5):
self.image_height = image_height
self.image_width = image_width
self.scale = scale
self.sigma = sigma
def generate_heatmap(self, points):
# points = (num_pts, 2)
h, w = self.image_height, self.image_width
pointmaps = []
for i in range(len(points)):
pointmap = np.zeros([h, w], dtype=np.float32)
# align_corners: False.
point = copy.deepcopy(points[i])
point[0] = max(0, min(w - 1, point[0]))
point[1] = max(0, min(h - 1, point[1]))
pointmap = self._circle(pointmap, point, sigma=self.sigma)
pointmaps.append(pointmap)
pointmaps = np.stack(pointmaps, axis=0) / 255.0
pointmaps = torch.from_numpy(pointmaps).float().unsqueeze(0)
pointmaps = F.interpolate(
pointmaps,
size=(int(w * self.scale), int(h * self.scale)),
mode='bilinear',
align_corners=False).squeeze()
return pointmaps
def _circle(self, img, pt, sigma=1.0, label_type='Gaussian'):
# Check that any part of the gaussian is in-bounds
tmp_size = sigma * 3
ul = [int(pt[0] - tmp_size), int(pt[1] - tmp_size)]
br = [int(pt[0] + tmp_size + 1), int(pt[1] + tmp_size + 1)]
if (ul[0] > img.shape[1] - 1 or ul[1] > img.shape[0] - 1
or br[0] - 1 < 0 or br[1] - 1 < 0):
# If not, just return the image as is
return img
# Generate gaussian
size = 2 * tmp_size + 1
x = np.arange(0, size, 1, np.float32)
y = x[:, np.newaxis]
x0 = y0 = size // 2
# The gaussian is not normalized, we want the center value to equal 1
if label_type == 'Gaussian':
g = np.exp(-((x - x0)**2 + (y - y0)**2) / (2 * sigma**2))
else:
g = sigma / (((x - x0)**2 + (y - y0)**2 + sigma**2)**1.5)
# Usable gaussian range
g_x = max(0, -ul[0]), min(br[0], img.shape[1]) - ul[0]
g_y = max(0, -ul[1]), min(br[1], img.shape[0]) - ul[1]
# Image range
img_x = max(0, ul[0]), min(br[0], img.shape[1])
img_y = max(0, ul[1]), min(br[1], img.shape[0])
img[img_y[0]:img_y[1],
img_x[0]:img_x[1]] = 255 * g[g_y[0]:g_y[1], g_x[0]:g_x[1]]
return img

View File

@@ -1,52 +1,54 @@
import json
import os.path as osp
import time
import torch
import numpy as np
# private package
from ..conf import *
from .backbone import StackedHGNetV1
def get_config(args):
config = None
config_name = args.config_name
if config_name == "alignment":
config = Alignment(args)
else:
assert NotImplementedError
return config
def get_net(config):
net = None
if config.net == "stackedHGnet_v1":
net = StackedHGNetV1(config=config,
classes_num=config.classes_num,
edge_info=config.edge_info,
nstack=config.nstack,
add_coord=config.add_coord,
decoder_type=config.decoder_type)
else:
assert False
return net
def set_environment(config):
if config.device_id >= 0:
assert torch.cuda.is_available() and torch.cuda.device_count() > config.device_id
torch.cuda.empty_cache()
config.device = torch.device("cuda", config.device_id)
config.use_gpu = True
else:
config.device = torch.device("cpu")
config.use_gpu = False
torch.set_default_dtype(torch.float32)
torch.set_default_tensor_type(torch.FloatTensor)
torch.set_flush_denormal(True) # ignore extremely small value
torch.backends.cudnn.benchmark = True # This flag allows you to enable the inbuilt cudnn auto-tuner to find the best algorithm to use for your hardware.
torch.autograd.set_detect_anomaly(True)
import os.path as osp
import time
import json
import numpy as np
import torch
from ..conf import *
from .backbone import StackedHGNetV1
def get_config(args):
config = None
config_name = args.config_name
if config_name == 'alignment':
config = Alignment(args)
else:
assert NotImplementedError
return config
def get_net(config):
net = None
if config.net == 'stackedHGnet_v1':
net = StackedHGNetV1(
config=config,
classes_num=config.classes_num,
edge_info=config.edge_info,
nstack=config.nstack,
add_coord=config.add_coord,
decoder_type=config.decoder_type)
else:
assert False
return net
def set_environment(config):
if config.device_id >= 0:
assert torch.cuda.is_available(
) and torch.cuda.device_count() > config.device_id
torch.cuda.empty_cache()
config.device = torch.device('cuda', config.device_id)
config.use_gpu = True
else:
config.device = torch.device('cpu')
config.use_gpu = False
torch.set_default_dtype(torch.float32)
torch.set_default_tensor_type(torch.FloatTensor)
torch.set_flush_denormal(True) # ignore extremely small value
torch.backends.cudnn.benchmark = True
# This flag allows you to enable the inbuilt cudnn auto-tuner to find the best algorithm to use for your hardware.
torch.autograd.set_detect_anomaly(True)

View File

@@ -1,34 +1,35 @@
# Copyright (c) Alibaba, Inc. and its affiliates.
import os
import numpy as np
import torch
import cv2
import matplotlib.pyplot as plt
import numpy as np
import torch
from modelscope.metainfo import Models
from modelscope.models.base.base_torch_model import TorchModel
from modelscope.models.builder import MODELS
from modelscope.preprocessors import LoadImage
from modelscope.models.cv.facial_68ldk_detection import infer
from modelscope.models.cv.facial_68ldk_detection import infer
from modelscope.outputs import OutputKeys
from modelscope.preprocessors import LoadImage
from modelscope.utils.constant import ModelFile, Tasks
from modelscope.utils.logger import get_logger
logger = get_logger()
@MODELS.register_module(
Tasks.facial_68ldk_detection, module_name=Models.star_68ldk_detection)
class FaceLandmarkDetection(TorchModel):
def __init__(self, model_dir, *args, **kwargs):
super().__init__(model_dir, *args, **kwargs)
def forward(self, Inputs):
return Inputs
def postprocess(self, Inputs):
return Inputs
def inference(self, data):
return data
return data

View File

@@ -1,25 +1,24 @@
# Copyright (c) Alibaba, Inc. and its affiliates.
from typing import Any, Dict, Union
import numpy as np
import torch
import cv2
import argparse
import os
from typing import Any, Dict, Union
import cv2
import numpy as np
import torch
from modelscope.metainfo import Pipelines
from modelscope.models.cv.facial_68ldk_detection import infer
from modelscope.outputs import OutputKeys
from modelscope.pipelines.base import Input, Model, Pipeline
from modelscope.pipelines.builder import PIPELINES
from modelscope.preprocessors import LoadImage
from modelscope.models.cv.facial_68ldk_detection import infer
from modelscope.outputs import OutputKeys
from modelscope.utils.constant import ModelFile, Tasks
from modelscope.utils.logger import get_logger
logger = get_logger()
@PIPELINES.register_module(
Tasks.facial_68ldk_detection, module_name=Pipelines.facial_68ldk_detection)
class FaceLandmarkDetectionPipeline(Pipeline):
@@ -32,37 +31,38 @@ class FaceLandmarkDetectionPipeline(Pipeline):
"""
super().__init__(model=model, **kwargs)
parser = argparse.ArgumentParser(description="Evaluation script")
parser = argparse.ArgumentParser(description='Evaluation script')
args = parser.parse_args()
args.config_name = 'alignment'
device_ids = list()
if torch.cuda.is_available():
device_ids = [0]
else:
else:
device_ids = [-1]
model_path = os.path.join(model, 'pytorch_model.pkl')
self.fld = infer.Alignment(args, model_path, dl_framework="pytorch", device_ids=device_ids)
self.fld = infer.Alignment(
args, model_path, dl_framework='pytorch', device_ids=device_ids)
logger.info('Face 2d landmark detection model, pipeline init')
def preprocess(self, input: Input) -> Dict[str, Any]:
print('start preprocess')
image = LoadImage.convert_to_ndarray(input)
image = cv2.resize(image, (256, 256))
data = {'image': image}
print('finish preprocess')
return data
def forward(self, input: Dict[str, Any]) -> Dict[str, Any]:
print('start infer')
image = input['image']
if torch.cuda.is_available():
@@ -71,15 +71,16 @@ class FaceLandmarkDetectionPipeline(Pipeline):
image_np = image.numpy()
x1, y1, x2, y2 = 0, 0, 256, 256
scale = max(x2 - x1, y2 - y1) / 180
scale = max(x2 - x1, y2 - y1) / 180
center_w = (x1 + x2) / 2
center_h = (y1 + y2) / 2
scale, center_w, center_h = float(scale), float(center_w), float(center_h)
scale, center_w, center_h = float(scale), float(center_w), float(
center_h)
results = self.fld.analyze(image_np, scale, center_w, center_h)
print('finish infer')
return results
def postprocess(self, inputs: Dict[str, Any]) -> Dict[str, Any]:

View File

@@ -46,10 +46,10 @@ class HFUtilTest(unittest.TestCase):
def test_transformer_patch(self):
tokenizer = AutoTokenizer.from_pretrained(
'skyline2006/llama-7b', revision='v1.0.1')
'iic/nlp_structbert_sentiment-classification_chinese-base')
self.assertIsNotNone(tokenizer)
model = AutoModelForCausalLM.from_pretrained(
'skyline2006/llama-7b', revision='v1.0.1')
'iic/nlp_structbert_sentiment-classification_chinese-base')
self.assertIsNotNone(model)