diff --git a/modelscope/metainfo.py b/modelscope/metainfo.py index c9ce5cb7..2c2128d8 100644 --- a/modelscope/metainfo.py +++ b/modelscope/metainfo.py @@ -39,7 +39,7 @@ class Models(object): body_3d_keypoints_hdformer = 'hdformer' crowd_counting = 'HRNetCrowdCounting' face_2d_keypoints = 'face-2d-keypoints' - star_68ldk_detection = 'star-68ldk-detection' + star_68ldk_detection = 'star-68ldk-detection' panoptic_segmentation = 'swinL-panoptic-segmentation' r50_panoptic_segmentation = 'r50-panoptic-segmentation' image_reid_person = 'passvitb' diff --git a/modelscope/models/cv/facial_68ldk_detection/conf/__init__.py b/modelscope/models/cv/facial_68ldk_detection/conf/__init__.py index 2f92d0e8..4690762b 100644 --- a/modelscope/models/cv/facial_68ldk_detection/conf/__init__.py +++ b/modelscope/models/cv/facial_68ldk_detection/conf/__init__.py @@ -1 +1 @@ -from .alignment import Alignment \ No newline at end of file +from .alignment import Alignment diff --git a/modelscope/models/cv/facial_68ldk_detection/conf/alignment.py b/modelscope/models/cv/facial_68ldk_detection/conf/alignment.py index eebaa1d7..30b5773d 100644 --- a/modelscope/models/cv/facial_68ldk_detection/conf/alignment.py +++ b/modelscope/models/cv/facial_68ldk_detection/conf/alignment.py @@ -1,239 +1,353 @@ -import os.path as osp -from .base import Base - - -class Alignment(Base): - """ - Alignment configure file, which contains training parameters of alignment. - """ - - def __init__(self, args): - super(Alignment, self).__init__('alignment') - self.ckpt_dir = '/mnt/workspace/humanAIGC/project/STAR/weights' - self.net = "stackedHGnet_v1" - self.nstack = 4 - self.loader_type = "alignment" - self.data_definition = "300W" # COFW, 300W, WFLW - self.test_file = "test.tsv" - - # image - self.channels = 3 - self.width = 256 - self.height = 256 - self.means = (127.5, 127.5, 127.5) - self.scale = 1 / 127.5 - self.aug_prob = 1.0 - - self.display_iteration = 10 - self.val_epoch = 1 - self.valset = "test.tsv" - self.norm_type = 'default' - self.encoder_type = 'default' - self.decoder_type = 'default' - - # scheduler & optimizer - self.milestones = [200, 350, 450] - self.max_epoch = 260 - self.optimizer = "adam" - self.learn_rate = 0.001 - self.weight_decay = 0.00001 - self.betas = [0.9, 0.999] - self.gamma = 0.1 - - # batch_size & workers - self.batch_size = 32 - self.train_num_workers = 16 - self.val_batch_size = 32 - self.val_num_workers = 16 - self.test_batch_size = 16 - self.test_num_workers = 0 - - # tricks - self.ema = True - self.add_coord = True - self.use_AAM = True - - # loss - self.loss_func = "STARLoss_v2" - - # STAR Loss paras - self.star_w = 1 - self.star_dist = 'smoothl1' - - self.init_from_args(args) - - # COFW - if self.data_definition == "COFW": - self.edge_info = ( - (True, (0, 4, 2, 5)), # RightEyebrow - (True, (1, 6, 3, 7)), # LeftEyebrow - (True, (8, 12, 10, 13)), # RightEye - (False, (9, 14, 11, 15)), # LeftEye - (True, (18, 20, 19, 21)), # Nose - (True, (22, 26, 23, 27)), # LowerLip - (True, (22, 24, 23, 25)), # UpperLip - ) - if self.norm_type == 'ocular': - self.nme_left_index = 8 # ocular - self.nme_right_index = 9 # ocular - elif self.norm_type in ['pupil', 'default']: - self.nme_left_index = 16 # pupil - self.nme_right_index = 17 # pupil - else: - raise NotImplementedError - self.classes_num = [29, 7, 29] - self.crop_op = True - self.flip_mapping = ( - [0, 1], [4, 6], [2, 3], [5, 7], [8, 9], [10, 11], [12, 14], [16, 17], [13, 15], [18, 19], [22, 23], - ) - self.image_dir = osp.join(self.image_dir, 'COFW') - # 300W - elif self.data_definition == "300W": - self.edge_info = ( - (False, (0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16)), # FaceContour - (False, (17, 18, 19, 20, 21)), # RightEyebrow - (False, (22, 23, 24, 25, 26)), # LeftEyebrow - (False, (27, 28, 29, 30)), # NoseLine - (False, (31, 32, 33, 34, 35)), # Nose - (True, (36, 37, 38, 39, 40, 41)), # RightEye - (True, (42, 43, 44, 45, 46, 47)), # LeftEye - (True, (48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 58, 59)), # OuterLip - (True, (60, 61, 62, 63, 64, 65, 66, 67)), # InnerLip - ) - if self.norm_type in ['ocular', 'default']: - self.nme_left_index = 36 # ocular - self.nme_right_index = 45 # ocular - elif self.norm_type == 'pupil': - self.nme_left_index = [36, 37, 38, 39, 40, 41] # pupil - self.nme_right_index = [42, 43, 44, 45, 46, 47] # pupil - else: - raise NotImplementedError - self.classes_num = [68, 9, 68] - self.crop_op = True - self.flip_mapping = ( - [0, 16], [1, 15], [2, 14], [3, 13], [4, 12], [5, 11], [6, 10], [7, 9], - [17, 26], [18, 25], [19, 24], [20, 23], [21, 22], - [31, 35], [32, 34], - [36, 45], [37, 44], [38, 43], [39, 42], [40, 47], [41, 46], - [48, 54], [49, 53], [50, 52], [61, 63], [60, 64], [67, 65], [58, 56], [59, 55], - ) - self.image_dir = osp.join(self.image_dir, '300W') - # self.image_dir = osp.join(self.image_dir, '300VW_images') - # 300VW - elif self.data_definition == "300VW": - self.edge_info = ( - (False, (0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16)), # FaceContour - (False, (17, 18, 19, 20, 21)), # RightEyebrow - (False, (22, 23, 24, 25, 26)), # LeftEyebrow - (False, (27, 28, 29, 30)), # NoseLine - (False, (31, 32, 33, 34, 35)), # Nose - (True, (36, 37, 38, 39, 40, 41)), # RightEye - (True, (42, 43, 44, 45, 46, 47)), # LeftEye - (True, (48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 58, 59)), # OuterLip - (True, (60, 61, 62, 63, 64, 65, 66, 67)), # InnerLip - ) - if self.norm_type in ['ocular', 'default']: - self.nme_left_index = 36 # ocular - self.nme_right_index = 45 # ocular - elif self.norm_type == 'pupil': - self.nme_left_index = [36, 37, 38, 39, 40, 41] # pupil - self.nme_right_index = [42, 43, 44, 45, 46, 47] # pupil - else: - raise NotImplementedError - self.classes_num = [68, 9, 68] - self.crop_op = True - self.flip_mapping = ( - [0, 16], [1, 15], [2, 14], [3, 13], [4, 12], [5, 11], [6, 10], [7, 9], - [17, 26], [18, 25], [19, 24], [20, 23], [21, 22], - [31, 35], [32, 34], - [36, 45], [37, 44], [38, 43], [39, 42], [40, 47], [41, 46], - [48, 54], [49, 53], [50, 52], [61, 63], [60, 64], [67, 65], [58, 56], [59, 55], - ) - self.image_dir = osp.join(self.image_dir, '300VW_Dataset_2015_12_14') - # WFLW - elif self.data_definition == "WFLW": - self.edge_info = ( - (False, ( - 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, - 27, - 28, 29, 30, 31, 32)), # FaceContour - (True, (33, 34, 35, 36, 37, 38, 39, 40, 41)), # RightEyebrow - (True, (42, 43, 44, 45, 46, 47, 48, 49, 50)), # LeftEyebrow - (False, (51, 52, 53, 54)), # NoseLine - (False, (55, 56, 57, 58, 59)), # Nose - (True, (60, 61, 62, 63, 64, 65, 66, 67)), # RightEye - (True, (68, 69, 70, 71, 72, 73, 74, 75)), # LeftEye - (True, (76, 77, 78, 79, 80, 81, 82, 83, 84, 85, 86, 87)), # OuterLip - (True, (88, 89, 90, 91, 92, 93, 94, 95)), # InnerLip - ) - if self.norm_type in ['ocular', 'default']: - self.nme_left_index = 60 # ocular - self.nme_right_index = 72 # ocular - elif self.norm_type == 'pupil': - self.nme_left_index = 96 # pupils - self.nme_right_index = 97 # pupils - else: - raise NotImplementedError - self.classes_num = [98, 9, 98] - self.crop_op = True - self.flip_mapping = ( - [0, 32], [1, 31], [2, 30], [3, 29], [4, 28], [5, 27], [6, 26], [7, 25], [8, 24], [9, 23], [10, 22], - [11, 21], [12, 20], [13, 19], [14, 18], [15, 17], # cheek - [33, 46], [34, 45], [35, 44], [36, 43], [37, 42], [38, 50], [39, 49], [40, 48], [41, 47], # elbrow - [60, 72], [61, 71], [62, 70], [63, 69], [64, 68], [65, 75], [66, 74], [67, 73], - [55, 59], [56, 58], - [76, 82], [77, 81], [78, 80], [87, 83], [86, 84], - [88, 92], [89, 91], [95, 93], [96, 97] - ) - self.image_dir = osp.join(self.image_dir, 'WFLW', 'WFLW_images') - - self.label_num = self.nstack * 3 if self.use_AAM else self.nstack - self.loss_weights, self.criterions, self.metrics = [], [], [] - for i in range(self.nstack): - factor = (2 ** i) / (2 ** (self.nstack - 1)) - if self.use_AAM: - self.loss_weights += [factor * weight for weight in [1.0, 10.0, 10.0]] - self.criterions += [self.loss_func, "AWingLoss", "AWingLoss"] - self.metrics += ["NME", None, None] - else: - self.loss_weights += [factor * weight for weight in [1.0]] - self.criterions += [self.loss_func, ] - self.metrics += ["NME", ] - - self.key_metric_index = (self.nstack - 1) * 3 if self.use_AAM else (self.nstack - 1) - - # data - self.folder = self.get_foldername() - self.work_dir = osp.join(self.ckpt_dir, self.data_definition, self.folder) - self.model_dir = osp.join(self.work_dir, 'model') - self.log_dir = osp.join(self.work_dir, 'log') - - self.train_tsv_file = osp.join(self.annot_dir, self.data_definition, "train.tsv") - self.train_pic_dir = self.image_dir - - self.val_tsv_file = osp.join(self.annot_dir, self.data_definition, self.valset) - self.val_pic_dir = self.image_dir - - self.test_tsv_file = osp.join(self.annot_dir, self.data_definition, self.test_file) - self.test_pic_dir = self.image_dir - - # self.train_tsv_file = osp.join(self.annot_dir, '300VW', "train.tsv") - # self.train_pic_dir = self.image_dir - - # self.val_tsv_file = osp.join(self.annot_dir, '300VW', self.valset) - # self.val_pic_dir = self.image_dir - - # self.test_tsv_file = osp.join(self.annot_dir, '300VW', self.test_file) - # self.test_pic_dir = self.image_dir - - - def get_foldername(self): - str = '' - str += '{}_{}x{}_{}_ep{}_lr{}_bs{}'.format(self.data_definition, self.height, self.width, - self.optimizer, self.max_epoch, self.learn_rate, self.batch_size) - str += '_{}'.format(self.loss_func) - str += '_{}_{}'.format(self.star_dist, self.star_w) if self.loss_func == 'STARLoss' else '' - str += '_AAM' if self.use_AAM else '' - str += '_{}'.format(self.valset[:-4]) if self.valset != 'test.tsv' else '' - str += '_{}'.format(self.id) - return str +import os.path as osp + +from .base import Base + + +class Alignment(Base): + """ + Alignment configure file, which contains training parameters of alignment. + """ + + def __init__(self, args): + super(Alignment, self).__init__('alignment') + self.ckpt_dir = '/mnt/workspace/humanAIGC/project/STAR/weights' + self.net = 'stackedHGnet_v1' + self.nstack = 4 + self.loader_type = 'alignment' + self.data_definition = '300W' # COFW, 300W, WFLW + self.test_file = 'test.tsv' + + # image + self.channels = 3 + self.width = 256 + self.height = 256 + self.means = (127.5, 127.5, 127.5) + self.scale = 1 / 127.5 + self.aug_prob = 1.0 + + self.display_iteration = 10 + self.val_epoch = 1 + self.valset = 'test.tsv' + self.norm_type = 'default' + self.encoder_type = 'default' + self.decoder_type = 'default' + + # scheduler & optimizer + self.milestones = [200, 350, 450] + self.max_epoch = 260 + self.optimizer = 'adam' + self.learn_rate = 0.001 + self.weight_decay = 0.00001 + self.betas = [0.9, 0.999] + self.gamma = 0.1 + + # batch_size & workers + self.batch_size = 32 + self.train_num_workers = 16 + self.val_batch_size = 32 + self.val_num_workers = 16 + self.test_batch_size = 16 + self.test_num_workers = 0 + + # tricks + self.ema = True + self.add_coord = True + self.use_AAM = True + + # loss + self.loss_func = 'STARLoss_v2' + + # STAR Loss paras + self.star_w = 1 + self.star_dist = 'smoothl1' + + self.init_from_args(args) + + # COFW + if self.data_definition == 'COFW': + self.edge_info = ( + (True, (0, 4, 2, 5)), # RightEyebrow + (True, (1, 6, 3, 7)), # LeftEyebrow + (True, (8, 12, 10, 13)), # RightEye + (False, (9, 14, 11, 15)), # LeftEye + (True, (18, 20, 19, 21)), # Nose + (True, (22, 26, 23, 27)), # LowerLip + (True, (22, 24, 23, 25)), # UpperLip + ) + if self.norm_type == 'ocular': + self.nme_left_index = 8 # ocular + self.nme_right_index = 9 # ocular + elif self.norm_type in ['pupil', 'default']: + self.nme_left_index = 16 # pupil + self.nme_right_index = 17 # pupil + else: + raise NotImplementedError + self.classes_num = [29, 7, 29] + self.crop_op = True + self.flip_mapping = ( + [0, 1], + [4, 6], + [2, 3], + [5, 7], + [8, 9], + [10, 11], + [12, 14], + [16, 17], + [13, 15], + [18, 19], + [22, 23], + ) + self.image_dir = osp.join(self.image_dir, 'COFW') + # 300W + elif self.data_definition == '300W': + self.edge_info = ( + (False, (0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, + 16)), # FaceContour + (False, (17, 18, 19, 20, 21)), # RightEyebrow + (False, (22, 23, 24, 25, 26)), # LeftEyebrow + (False, (27, 28, 29, 30)), # NoseLine + (False, (31, 32, 33, 34, 35)), # Nose + (True, (36, 37, 38, 39, 40, 41)), # RightEye + (True, (42, 43, 44, 45, 46, 47)), # LeftEye + (True, (48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 58, + 59)), # OuterLip + (True, (60, 61, 62, 63, 64, 65, 66, 67)), # InnerLip + ) + if self.norm_type in ['ocular', 'default']: + self.nme_left_index = 36 # ocular + self.nme_right_index = 45 # ocular + elif self.norm_type == 'pupil': + self.nme_left_index = [36, 37, 38, 39, 40, 41] # pupil + self.nme_right_index = [42, 43, 44, 45, 46, 47] # pupil + else: + raise NotImplementedError + self.classes_num = [68, 9, 68] + self.crop_op = True + self.flip_mapping = ( + [0, 16], + [1, 15], + [2, 14], + [3, 13], + [4, 12], + [5, 11], + [6, 10], + [7, 9], + [17, 26], + [18, 25], + [19, 24], + [20, 23], + [21, 22], + [31, 35], + [32, 34], + [36, 45], + [37, 44], + [38, 43], + [39, 42], + [40, 47], + [41, 46], + [48, 54], + [49, 53], + [50, 52], + [61, 63], + [60, 64], + [67, 65], + [58, 56], + [59, 55], + ) + self.image_dir = osp.join(self.image_dir, '300W') + # self.image_dir = osp.join(self.image_dir, '300VW_images') + # 300VW + elif self.data_definition == '300VW': + self.edge_info = ( + (False, (0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, + 16)), # FaceContour + (False, (17, 18, 19, 20, 21)), # RightEyebrow + (False, (22, 23, 24, 25, 26)), # LeftEyebrow + (False, (27, 28, 29, 30)), # NoseLine + (False, (31, 32, 33, 34, 35)), # Nose + (True, (36, 37, 38, 39, 40, 41)), # RightEye + (True, (42, 43, 44, 45, 46, 47)), # LeftEye + (True, (48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 58, + 59)), # OuterLip + (True, (60, 61, 62, 63, 64, 65, 66, 67)), # InnerLip + ) + if self.norm_type in ['ocular', 'default']: + self.nme_left_index = 36 # ocular + self.nme_right_index = 45 # ocular + elif self.norm_type == 'pupil': + self.nme_left_index = [36, 37, 38, 39, 40, 41] # pupil + self.nme_right_index = [42, 43, 44, 45, 46, 47] # pupil + else: + raise NotImplementedError + self.classes_num = [68, 9, 68] + self.crop_op = True + self.flip_mapping = ( + [0, 16], + [1, 15], + [2, 14], + [3, 13], + [4, 12], + [5, 11], + [6, 10], + [7, 9], + [17, 26], + [18, 25], + [19, 24], + [20, 23], + [21, 22], + [31, 35], + [32, 34], + [36, 45], + [37, 44], + [38, 43], + [39, 42], + [40, 47], + [41, 46], + [48, 54], + [49, 53], + [50, 52], + [61, 63], + [60, 64], + [67, 65], + [58, 56], + [59, 55], + ) + self.image_dir = osp.join(self.image_dir, + '300VW_Dataset_2015_12_14') + # WFLW + elif self.data_definition == 'WFLW': + self.edge_info = ( + (False, (0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, + 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, + 29, 30, 31, 32)), # FaceContour + (True, (33, 34, 35, 36, 37, 38, 39, 40, 41)), # RightEyebrow + (True, (42, 43, 44, 45, 46, 47, 48, 49, 50)), # LeftEyebrow + (False, (51, 52, 53, 54)), # NoseLine + (False, (55, 56, 57, 58, 59)), # Nose + (True, (60, 61, 62, 63, 64, 65, 66, 67)), # RightEye + (True, (68, 69, 70, 71, 72, 73, 74, 75)), # LeftEye + (True, (76, 77, 78, 79, 80, 81, 82, 83, 84, 85, 86, + 87)), # OuterLip + (True, (88, 89, 90, 91, 92, 93, 94, 95)), # InnerLip + ) + if self.norm_type in ['ocular', 'default']: + self.nme_left_index = 60 # ocular + self.nme_right_index = 72 # ocular + elif self.norm_type == 'pupil': + self.nme_left_index = 96 # pupils + self.nme_right_index = 97 # pupils + else: + raise NotImplementedError + self.classes_num = [98, 9, 98] + self.crop_op = True + self.flip_mapping = ( + [0, 32], + [1, 31], + [2, 30], + [3, 29], + [4, 28], + [5, 27], + [6, 26], + [7, 25], + [8, 24], + [9, 23], + [10, 22], + [11, 21], + [12, 20], + [13, 19], + [14, 18], + [15, 17], # cheek + [33, 46], + [34, 45], + [35, 44], + [36, 43], + [37, 42], + [38, 50], + [39, 49], + [40, 48], + [41, 47], # elbrow + [60, 72], + [61, 71], + [62, 70], + [63, 69], + [64, 68], + [65, 75], + [66, 74], + [67, 73], + [55, 59], + [56, 58], + [76, 82], + [77, 81], + [78, 80], + [87, 83], + [86, 84], + [88, 92], + [89, 91], + [95, 93], + [96, 97]) + self.image_dir = osp.join(self.image_dir, 'WFLW', 'WFLW_images') + + self.label_num = self.nstack * 3 if self.use_AAM else self.nstack + self.loss_weights, self.criterions, self.metrics = [], [], [] + for i in range(self.nstack): + factor = (2**i) / (2**(self.nstack - 1)) + if self.use_AAM: + self.loss_weights += [ + factor * weight for weight in [1.0, 10.0, 10.0] + ] + self.criterions += [self.loss_func, 'AWingLoss', 'AWingLoss'] + self.metrics += ['NME', None, None] + else: + self.loss_weights += [factor * weight for weight in [1.0]] + self.criterions += [ + self.loss_func, + ] + self.metrics += [ + 'NME', + ] + + self.key_metric_index = (self.nstack - 1) * 3 if self.use_AAM else ( + self.nstack - 1) + + # data + self.folder = self.get_foldername() + self.work_dir = osp.join(self.ckpt_dir, self.data_definition, + self.folder) + self.model_dir = osp.join(self.work_dir, 'model') + self.log_dir = osp.join(self.work_dir, 'log') + + self.train_tsv_file = osp.join(self.annot_dir, self.data_definition, + 'train.tsv') + self.train_pic_dir = self.image_dir + + self.val_tsv_file = osp.join(self.annot_dir, self.data_definition, + self.valset) + self.val_pic_dir = self.image_dir + + self.test_tsv_file = osp.join(self.annot_dir, self.data_definition, + self.test_file) + self.test_pic_dir = self.image_dir + + # self.train_tsv_file = osp.join(self.annot_dir, '300VW', "train.tsv") + # self.train_pic_dir = self.image_dir + + # self.val_tsv_file = osp.join(self.annot_dir, '300VW', self.valset) + # self.val_pic_dir = self.image_dir + + # self.test_tsv_file = osp.join(self.annot_dir, '300VW', self.test_file) + # self.test_pic_dir = self.image_dir + + def get_foldername(self): + str = '' + str += '{}_{}x{}_{}_ep{}_lr{}_bs{}'.format( + self.data_definition, self.height, self.width, self.optimizer, + self.max_epoch, self.learn_rate, self.batch_size) + str += '_{}'.format(self.loss_func) + str += '_{}_{}'.format( + self.star_dist, + self.star_w) if self.loss_func == 'STARLoss' else '' + str += '_AAM' if self.use_AAM else '' + str += '_{}'.format( + self.valset[:-4]) if self.valset != 'test.tsv' else '' + str += '_{}'.format(self.id) + return str diff --git a/modelscope/models/cv/facial_68ldk_detection/conf/base.py b/modelscope/models/cv/facial_68ldk_detection/conf/base.py index 55aded09..30450524 100644 --- a/modelscope/models/cv/facial_68ldk_detection/conf/base.py +++ b/modelscope/models/cv/facial_68ldk_detection/conf/base.py @@ -1,94 +1,102 @@ -import uuid -import logging -import os.path as osp -from argparse import Namespace -# from tensorboardX import SummaryWriter - -class Base: - """ - Base configure file, which contains the basic training parameters and should be inherited by other attribute configure file. - """ - - def __init__(self, config_name, ckpt_dir='./', image_dir='./', annot_dir='./'): - self.type = config_name - self.id = str(uuid.uuid4()) - self.note = "" - - self.ckpt_dir = ckpt_dir - self.image_dir = image_dir - self.annot_dir = annot_dir - - self.loader_type = "alignment" - self.loss_func = "STARLoss" - - # train - self.batch_size = 128 - self.val_batch_size = 1 - self.test_batch_size = 32 - self.channels = 3 - self.width = 256 - self.height = 256 - - # mean values in r, g, b channel. - self.means = (127, 127, 127) - self.scale = 0.0078125 - - self.display_iteration = 100 - self.milestones = [50, 80] - self.max_epoch = 100 - - self.net = "stackedHGnet_v1" - self.nstack = 4 - - # ["adam", "sgd"] - self.optimizer = "adam" - self.learn_rate = 0.1 - self.momentum = 0.01 # caffe: 0.99 - self.weight_decay = 0.0 - self.nesterov = False - self.scheduler = "MultiStepLR" - self.gamma = 0.1 - - self.loss_weights = [1.0] - self.criterions = ["SoftmaxWithLoss"] - self.metrics = ["Accuracy"] - self.key_metric_index = 0 - self.classes_num = [1000] - self.label_num = len(self.classes_num) - - # model - self.ema = False - self.use_AAM = True - - # visualization - self.writer = None - - # log file - self.logger = None - - def init_instance(self): - # self.writer = SummaryWriter(logdir=self.log_dir, comment=self.type) - log_formatter = logging.Formatter("%(asctime)s %(levelname)-8s: %(message)s") - root_logger = logging.getLogger() - file_handler = logging.FileHandler(osp.join(self.log_dir, "log.txt")) - file_handler.setFormatter(log_formatter) - file_handler.setLevel(logging.NOTSET) - root_logger.addHandler(file_handler) - console_handler = logging.StreamHandler() - console_handler.setFormatter(log_formatter) - console_handler.setLevel(logging.NOTSET) - root_logger.addHandler(console_handler) - root_logger.setLevel(logging.NOTSET) - self.logger = root_logger - - def __del__(self): - # tensorboard --logdir self.log_dir - if self.writer is not None: - # self.writer.export_scalars_to_json(self.log_dir + "visual.json") - self.writer.close() - - def init_from_args(self, args: Namespace): - args_vars = vars(args) - for key, value in args_vars.items(): - if hasattr(self, key) and value is not None: - setattr(self, key, value) +import logging +import os.path as osp +import uuid +from argparse import Namespace + +# from tensorboardX import SummaryWriter + + +class Base: + """ + Base configure file, which contains the basic training parameters + and should be inherited by other attribute configure file. + """ + + def __init__(self, + config_name, + ckpt_dir='./', + image_dir='./', + annot_dir='./'): + self.type = config_name + self.id = str(uuid.uuid4()) + self.note = '' + + self.ckpt_dir = ckpt_dir + self.image_dir = image_dir + self.annot_dir = annot_dir + + self.loader_type = 'alignment' + self.loss_func = 'STARLoss' + + # train + self.batch_size = 128 + self.val_batch_size = 1 + self.test_batch_size = 32 + self.channels = 3 + self.width = 256 + self.height = 256 + + # mean values in r, g, b channel. + self.means = (127, 127, 127) + self.scale = 0.0078125 + + self.display_iteration = 100 + self.milestones = [50, 80] + self.max_epoch = 100 + + self.net = 'stackedHGnet_v1' + self.nstack = 4 + + # ["adam", "sgd"] + self.optimizer = 'adam' + self.learn_rate = 0.1 + self.momentum = 0.01 # caffe: 0.99 + self.weight_decay = 0.0 + self.nesterov = False + self.scheduler = 'MultiStepLR' + self.gamma = 0.1 + + self.loss_weights = [1.0] + self.criterions = ['SoftmaxWithLoss'] + self.metrics = ['Accuracy'] + self.key_metric_index = 0 + self.classes_num = [1000] + self.label_num = len(self.classes_num) + + # model + self.ema = False + self.use_AAM = True + + # visualization + self.writer = None + + # log file + self.logger = None + + def init_instance(self): + # self.writer = SummaryWriter(logdir=self.log_dir, comment=self.type) + log_formatter = logging.Formatter( + '%(asctime)s %(levelname)-8s: %(message)s') + root_logger = logging.getLogger() + file_handler = logging.FileHandler(osp.join(self.log_dir, 'log.txt')) + file_handler.setFormatter(log_formatter) + file_handler.setLevel(logging.NOTSET) + root_logger.addHandler(file_handler) + console_handler = logging.StreamHandler() + console_handler.setFormatter(log_formatter) + console_handler.setLevel(logging.NOTSET) + root_logger.addHandler(console_handler) + root_logger.setLevel(logging.NOTSET) + self.logger = root_logger + + def __del__(self): + # tensorboard --logdir self.log_dir + if self.writer is not None: + # self.writer.export_scalars_to_json(self.log_dir + "visual.json") + self.writer.close() + + def init_from_args(self, args: Namespace): + args_vars = vars(args) + for key, value in args_vars.items(): + if hasattr(self, key) and value is not None: + setattr(self, key, value) diff --git a/modelscope/models/cv/facial_68ldk_detection/infer.py b/modelscope/models/cv/facial_68ldk_detection/infer.py index 597120f4..ccc6229a 100644 --- a/modelscope/models/cv/facial_68ldk_detection/infer.py +++ b/modelscope/models/cv/facial_68ldk_detection/infer.py @@ -1,13 +1,15 @@ -import cv2 -import math -import copy -import numpy as np import argparse +import copy +import math + +import cv2 +import numpy as np import torch # private package from .lib import utility + class GetCropMatrix(): """ from_shape -> transform_matrix @@ -18,7 +20,8 @@ class GetCropMatrix(): self.target_face_scale = target_face_scale self.align_corners = align_corners - def _compose_rotate_and_scale(self, angle, scale, shift_xy, from_center, to_center): + def _compose_rotate_and_scale(self, angle, scale, shift_xy, from_center, + to_center): cosv = math.cos(angle) sinv = math.sin(angle) @@ -36,11 +39,8 @@ class GetCropMatrix(): b1 = acos b2 = ty - asin * fx - acos * fy + shift_xy[1] - rot_scale_m = np.array([ - [a0, a1, a2], - [b0, b1, b2], - [0.0, 0.0, 1.0] - ], np.float32) + rot_scale_m = np.array([[a0, a1, a2], [b0, b1, b2], [0.0, 0.0, 1.0]], + np.float32) return rot_scale_m def process(self, scale, center_w, center_h): @@ -53,7 +53,9 @@ class GetCropMatrix(): scale_mu = self.image_size / (scale * self.target_face_scale * 200.0) shift_xy_mu = (0, 0) matrix = self._compose_rotate_and_scale( - rot_mu, scale_mu, shift_xy_mu, + rot_mu, + scale_mu, + shift_xy_mu, from_center=[center_w, center_h], to_center=[to_w / 2.0, to_h / 2.0]) return matrix @@ -69,8 +71,11 @@ class TransformPerspective(): def process(self, image, matrix): return cv2.warpPerspective( - image, matrix, dsize=(self.image_size, self.image_size), - flags=cv2.INTER_LINEAR, borderValue=0) + image, + matrix, + dsize=(self.image_size, self.image_size), + flags=cv2.INTER_LINEAR, + borderValue=0) class TransformPoints2D(): @@ -80,63 +85,78 @@ class TransformPoints2D(): def process(self, srcPoints, matrix): # nx3 - desPoints = np.concatenate([srcPoints, np.ones_like(srcPoints[:, [0]])], axis=1) + desPoints = np.concatenate( + [srcPoints, np.ones_like(srcPoints[:, [0]])], axis=1) desPoints = desPoints @ np.transpose(matrix) # nx3 desPoints = desPoints[:, :2] / desPoints[:, [2, 2]] return desPoints.astype(srcPoints.dtype) + class Alignment: + def __init__(self, args, model_path, dl_framework, device_ids): self.input_size = 256 self.target_face_scale = 1.0 self.dl_framework = dl_framework # model - if self.dl_framework == "pytorch": + if self.dl_framework == 'pytorch': # conf self.config = utility.get_config(args) self.config.device_id = device_ids[0] - + # set environment utility.set_environment(self.config) net = utility.get_net(self.config) if device_ids == [-1]: - checkpoint = torch.load(model_path, map_location="cpu") + checkpoint = torch.load(model_path, map_location='cpu') else: checkpoint = torch.load(model_path) - net.load_state_dict(checkpoint["net"]) + net.load_state_dict(checkpoint['net']) if self.config.device_id == -1: net = net.cpu() else: net = net.to(self.config.device_id) - + net.eval() self.alignment = net else: assert False - self.getCropMatrix = GetCropMatrix(image_size=self.input_size, target_face_scale=self.target_face_scale, - align_corners=True) - self.transformPerspective = TransformPerspective(image_size=self.input_size) + self.getCropMatrix = GetCropMatrix( + image_size=self.input_size, + target_face_scale=self.target_face_scale, + align_corners=True) + self.transformPerspective = TransformPerspective( + image_size=self.input_size) self.transformPoints2D = TransformPoints2D() def norm_points(self, points, align_corners=False): if align_corners: # [0, SIZE-1] -> [-1, +1] - return points / torch.tensor([self.input_size - 1, self.input_size - 1]).to(points).view(1, 1, 2) * 2 - 1 + return points / torch.tensor([ + self.input_size - 1, self.input_size - 1 + ]).to(points).view(1, 1, 2) * 2 - 1 else: # [-0.5, SIZE-0.5] -> [-1, +1] - return (points * 2 + 1) / torch.tensor([self.input_size, self.input_size]).to(points).view(1, 1, 2) - 1 + return (points * 2 + 1) / torch.tensor([ + self.input_size, self.input_size + ]).to(points).view(1, 1, 2) - 1 def denorm_points(self, points, align_corners=False): if align_corners: # [-1, +1] -> [0, SIZE-1] - return (points + 1) / 2 * torch.tensor([self.input_size - 1, self.input_size - 1]).to(points).view(1, 1, 2) + return (points + 1) / 2 * torch.tensor([ + self.input_size - 1, self.input_size - 1 + ]).to(points).view(1, 1, 2) else: # [-1, +1] -> [-0.5, SIZE-0.5] - return ((points + 1) * torch.tensor([self.input_size, self.input_size]).to(points).view(1, 1, 2) - 1) / 2 + return ((points + 1) * torch.tensor( # noqa + [self.input_size, self.input_size]).to(points).view(1, 1, + 2) # noqa + - 1) / 2 # noqa def preprocess(self, image, scale, center_w, center_h): matrix = self.getCropMatrix.process(scale, center_w, center_h) @@ -151,7 +171,7 @@ class Alignment: input_tensor = input_tensor.cpu() else: input_tensor = input_tensor.to(self.config.device_id) - + return input_tensor, matrix def postprocess(self, srcPoints, coeff): @@ -160,14 +180,17 @@ class Alignment: # src = matrix * dst dstPoints = np.zeros(srcPoints.shape, dtype=np.float32) for i in range(srcPoints.shape[0]): - dstPoints[i][0] = coeff[0][0] * srcPoints[i][0] + coeff[0][1] * srcPoints[i][1] + coeff[0][2] - dstPoints[i][1] = coeff[1][0] * srcPoints[i][0] + coeff[1][1] * srcPoints[i][1] + coeff[1][2] + dstPoints[i][0] = coeff[0][0] * srcPoints[i][0] + coeff[0][ + 1] * srcPoints[i][1] + coeff[0][2] + dstPoints[i][1] = coeff[1][0] * srcPoints[i][0] + coeff[1][ + 1] * srcPoints[i][1] + coeff[1][2] return dstPoints def analyze(self, image, scale, center_w, center_h): - input_tensor, matrix = self.preprocess(image, scale, center_w, center_h) + input_tensor, matrix = self.preprocess(image, scale, center_w, + center_h) - if self.dl_framework == "pytorch": + if self.dl_framework == 'pytorch': with torch.no_grad(): output = self.alignment(input_tensor) landmarks = output[-1][0] @@ -179,4 +202,3 @@ class Alignment: landmarks = self.postprocess(landmarks, np.linalg.inv(matrix)) return landmarks - diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/__init__.py b/modelscope/models/cv/facial_68ldk_detection/lib/__init__.py index 0f808518..a0efc10d 100644 --- a/modelscope/models/cv/facial_68ldk_detection/lib/__init__.py +++ b/modelscope/models/cv/facial_68ldk_detection/lib/__init__.py @@ -1,2 +1,2 @@ -from .backbone import StackedHGNetV1 -from .utility import get_config, get_net +from .backbone import StackedHGNetV1 +from .utility import get_config, get_net diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__init__.py b/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__init__.py index cb1578aa..5bbfc2e2 100644 --- a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__init__.py +++ b/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__init__.py @@ -1,5 +1,5 @@ -from .stackedHGNetV1 import StackedHGNetV1 - -__all__ = [ - "StackedHGNetV1", -] \ No newline at end of file +from .stackedHGNetV1 import StackedHGNetV1 + +__all__ = [ + 'StackedHGNetV1', +] diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__pycache__/__init__.cpython-312.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__pycache__/__init__.cpython-312.pyc deleted file mode 100644 index 91ae9bf0..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__pycache__/__init__.cpython-312.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__pycache__/__init__.cpython-37.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__pycache__/__init__.cpython-37.pyc deleted file mode 100644 index 8ec2c839..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__pycache__/__init__.cpython-37.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__pycache__/__init__.cpython-39.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__pycache__/__init__.cpython-39.pyc deleted file mode 100644 index a5cbbaba..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__pycache__/__init__.cpython-39.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__pycache__/stackedHGNetV1.cpython-312.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__pycache__/stackedHGNetV1.cpython-312.pyc deleted file mode 100644 index a47e8606..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__pycache__/stackedHGNetV1.cpython-312.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__pycache__/stackedHGNetV1.cpython-37.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__pycache__/stackedHGNetV1.cpython-37.pyc deleted file mode 100644 index 6737dfa6..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__pycache__/stackedHGNetV1.cpython-37.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__pycache__/stackedHGNetV1.cpython-39.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__pycache__/stackedHGNetV1.cpython-39.pyc deleted file mode 100644 index f69878f5..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/__pycache__/stackedHGNetV1.cpython-39.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/core/__pycache__/coord_conv.cpython-312.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/backbone/core/__pycache__/coord_conv.cpython-312.pyc deleted file mode 100644 index 6a178840..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/core/__pycache__/coord_conv.cpython-312.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/core/__pycache__/coord_conv.cpython-37.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/backbone/core/__pycache__/coord_conv.cpython-37.pyc deleted file mode 100644 index 35ecca22..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/core/__pycache__/coord_conv.cpython-37.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/core/__pycache__/coord_conv.cpython-39.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/backbone/core/__pycache__/coord_conv.cpython-39.pyc deleted file mode 100644 index e43d1ae1..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/core/__pycache__/coord_conv.cpython-39.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/core/coord_conv.py b/modelscope/models/cv/facial_68ldk_detection/lib/backbone/core/coord_conv.py index 7239421d..ca37ea55 100644 --- a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/core/coord_conv.py +++ b/modelscope/models/cv/facial_68ldk_detection/lib/backbone/core/coord_conv.py @@ -1,157 +1,187 @@ -import torch -import torch.nn as nn - - -class AddCoordsTh(nn.Module): - def __init__(self, x_dim, y_dim, with_r=False, with_boundary=False): - super(AddCoordsTh, self).__init__() - self.x_dim = x_dim - self.y_dim = y_dim - self.with_r = with_r - self.with_boundary = with_boundary - - def forward(self, input_tensor, heatmap=None): - """ - input_tensor: (batch, c, x_dim, y_dim) - """ - batch_size_tensor = input_tensor.shape[0] - - xx_ones = torch.ones([1, self.y_dim], dtype=torch.int32).to(input_tensor) - xx_ones = xx_ones.unsqueeze(-1) - - xx_range = torch.arange(self.x_dim, dtype=torch.int32).unsqueeze(0).to(input_tensor) - xx_range = xx_range.unsqueeze(1) - - xx_channel = torch.matmul(xx_ones.float(), xx_range.float()) - xx_channel = xx_channel.unsqueeze(-1) - - yy_ones = torch.ones([1, self.x_dim], dtype=torch.int32).to(input_tensor) - yy_ones = yy_ones.unsqueeze(1) - - yy_range = torch.arange(self.y_dim, dtype=torch.int32).unsqueeze(0).to(input_tensor) - yy_range = yy_range.unsqueeze(-1) - - yy_channel = torch.matmul(yy_range.float(), yy_ones.float()) - yy_channel = yy_channel.unsqueeze(-1) - - xx_channel = xx_channel.permute(0, 3, 2, 1) - yy_channel = yy_channel.permute(0, 3, 2, 1) - - xx_channel = xx_channel / (self.x_dim - 1) - yy_channel = yy_channel / (self.y_dim - 1) - - xx_channel = xx_channel * 2 - 1 - yy_channel = yy_channel * 2 - 1 - - xx_channel = xx_channel.repeat(batch_size_tensor, 1, 1, 1) - yy_channel = yy_channel.repeat(batch_size_tensor, 1, 1, 1) - - if self.with_boundary and type(heatmap) != type(None): - boundary_channel = torch.clamp(heatmap[:, -1:, :, :], - 0.0, 1.0) - - zero_tensor = torch.zeros_like(xx_channel).to(xx_channel) - xx_boundary_channel = torch.where(boundary_channel>0.05, - xx_channel, zero_tensor) - yy_boundary_channel = torch.where(boundary_channel>0.05, - yy_channel, zero_tensor) - ret = torch.cat([input_tensor, xx_channel, yy_channel], dim=1) - - - if self.with_r: - rr = torch.sqrt(torch.pow(xx_channel, 2) + torch.pow(yy_channel, 2)) - rr = rr / torch.max(rr) - ret = torch.cat([ret, rr], dim=1) - - if self.with_boundary and type(heatmap) != type(None): - ret = torch.cat([ret, xx_boundary_channel, - yy_boundary_channel], dim=1) - return ret - - -class CoordConvTh(nn.Module): - """CoordConv layer as in the paper.""" - def __init__(self, x_dim, y_dim, with_r, with_boundary, - in_channels, out_channels, first_one=False, relu=False, bn=False, *args, **kwargs): - super(CoordConvTh, self).__init__() - self.addcoords = AddCoordsTh(x_dim=x_dim, y_dim=y_dim, with_r=with_r, - with_boundary=with_boundary) - in_channels += 2 - if with_r: - in_channels += 1 - if with_boundary and not first_one: - in_channels += 2 - self.conv = nn.Conv2d(in_channels=in_channels, out_channels=out_channels, *args, **kwargs) - self.relu = nn.ReLU() if relu else None - self.bn = nn.BatchNorm2d(out_channels) if bn else None - - self.with_boundary = with_boundary - self.first_one = first_one - - - def forward(self, input_tensor, heatmap=None): - assert (self.with_boundary and not self.first_one) == (heatmap is not None) - ret = self.addcoords(input_tensor, heatmap) - ret = self.conv(ret) - if self.bn is not None: - ret = self.bn(ret) - if self.relu is not None: - ret = self.relu(ret) - - return ret - - -''' -An alternative implementation for PyTorch with auto-infering the x-y dimensions. -''' -class AddCoords(nn.Module): - - def __init__(self, with_r=False): - super().__init__() - self.with_r = with_r - - def forward(self, input_tensor): - """ - Args: - input_tensor: shape(batch, channel, x_dim, y_dim) - """ - batch_size, _, x_dim, y_dim = input_tensor.size() - - xx_channel = torch.arange(x_dim).repeat(1, y_dim, 1).to(input_tensor) - yy_channel = torch.arange(y_dim).repeat(1, x_dim, 1).transpose(1, 2).to(input_tensor) - - xx_channel = xx_channel / (x_dim - 1) - yy_channel = yy_channel / (y_dim - 1) - - xx_channel = xx_channel * 2 - 1 - yy_channel = yy_channel * 2 - 1 - - xx_channel = xx_channel.repeat(batch_size, 1, 1, 1).transpose(2, 3) - yy_channel = yy_channel.repeat(batch_size, 1, 1, 1).transpose(2, 3) - - ret = torch.cat([ - input_tensor, - xx_channel.type_as(input_tensor), - yy_channel.type_as(input_tensor)], dim=1) - - if self.with_r: - rr = torch.sqrt(torch.pow(xx_channel - 0.5, 2) + torch.pow(yy_channel - 0.5, 2)) - ret = torch.cat([ret, rr], dim=1) - - return ret - - -class CoordConv(nn.Module): - - def __init__(self, in_channels, out_channels, with_r=False, **kwargs): - super().__init__() - self.addcoords = AddCoords(with_r=with_r) - in_channels += 2 - if with_r: - in_channels += 1 - self.conv = nn.Conv2d(in_channels, out_channels, **kwargs) - - def forward(self, x): - ret = self.addcoords(x) - ret = self.conv(ret) - return ret +import torch +import torch.nn as nn + + +class AddCoordsTh(nn.Module): + + def __init__(self, x_dim, y_dim, with_r=False, with_boundary=False): + super(AddCoordsTh, self).__init__() + self.x_dim = x_dim + self.y_dim = y_dim + self.with_r = with_r + self.with_boundary = with_boundary + + def forward(self, input_tensor, heatmap=None): + """ + input_tensor: (batch, c, x_dim, y_dim) + """ + batch_size_tensor = input_tensor.shape[0] + + xx_ones = torch.ones([1, self.y_dim], + dtype=torch.int32).to(input_tensor) + xx_ones = xx_ones.unsqueeze(-1) + + xx_range = torch.arange( + self.x_dim, dtype=torch.int32).unsqueeze(0).to(input_tensor) + xx_range = xx_range.unsqueeze(1) + + xx_channel = torch.matmul(xx_ones.float(), xx_range.float()) + xx_channel = xx_channel.unsqueeze(-1) + + yy_ones = torch.ones([1, self.x_dim], + dtype=torch.int32).to(input_tensor) + yy_ones = yy_ones.unsqueeze(1) + + yy_range = torch.arange( + self.y_dim, dtype=torch.int32).unsqueeze(0).to(input_tensor) + yy_range = yy_range.unsqueeze(-1) + + yy_channel = torch.matmul(yy_range.float(), yy_ones.float()) + yy_channel = yy_channel.unsqueeze(-1) + + xx_channel = xx_channel.permute(0, 3, 2, 1) + yy_channel = yy_channel.permute(0, 3, 2, 1) + + xx_channel = xx_channel / (self.x_dim - 1) + yy_channel = yy_channel / (self.y_dim - 1) + + xx_channel = xx_channel * 2 - 1 + yy_channel = yy_channel * 2 - 1 + + xx_channel = xx_channel.repeat(batch_size_tensor, 1, 1, 1) + yy_channel = yy_channel.repeat(batch_size_tensor, 1, 1, 1) + + if self.with_boundary and heatmap is not None: + boundary_channel = torch.clamp(heatmap[:, -1:, :, :], 0.0, 1.0) + + zero_tensor = torch.zeros_like(xx_channel).to(xx_channel) + xx_boundary_channel = torch.where(boundary_channel > 0.05, + xx_channel, zero_tensor) + yy_boundary_channel = torch.where(boundary_channel > 0.05, + yy_channel, zero_tensor) + ret = torch.cat([input_tensor, xx_channel, yy_channel], dim=1) + + if self.with_r: + rr = torch.sqrt( + torch.pow(xx_channel, 2) + torch.pow(yy_channel, 2)) + rr = rr / torch.max(rr) + ret = torch.cat([ret, rr], dim=1) + + if self.with_boundary and heatmap is not None: + ret = torch.cat([ret, xx_boundary_channel, yy_boundary_channel], + dim=1) + return ret + + +class CoordConvTh(nn.Module): + """CoordConv layer as in the paper.""" + + def __init__(self, + x_dim, + y_dim, + with_r, + with_boundary, + in_channels, + out_channels, + first_one=False, + relu=False, + bn=False, + *args, + **kwargs): + super(CoordConvTh, self).__init__() + self.addcoords = AddCoordsTh( + x_dim=x_dim, + y_dim=y_dim, + with_r=with_r, + with_boundary=with_boundary) + in_channels += 2 + if with_r: + in_channels += 1 + if with_boundary and not first_one: + in_channels += 2 + self.conv = nn.Conv2d( + in_channels=in_channels, + out_channels=out_channels, + *args, + **kwargs) + self.relu = nn.ReLU() if relu else None + self.bn = nn.BatchNorm2d(out_channels) if bn else None + + self.with_boundary = with_boundary + self.first_one = first_one + + def forward(self, input_tensor, heatmap=None): + assert (self.with_boundary and not self.first_one) == ( + heatmap is not None) + ret = self.addcoords(input_tensor, heatmap) + ret = self.conv(ret) + if self.bn is not None: + ret = self.bn(ret) + if self.relu is not None: + ret = self.relu(ret) + + return ret + + +''' +An alternative implementation for PyTorch with auto-infering the x-y dimensions. +''' + + +class AddCoords(nn.Module): + + def __init__(self, with_r=False): + super().__init__() + self.with_r = with_r + + def forward(self, input_tensor): + """ + Args: + input_tensor: shape(batch, channel, x_dim, y_dim) + """ + batch_size, _, x_dim, y_dim = input_tensor.size() + + xx_channel = torch.arange(x_dim).repeat(1, y_dim, 1).to(input_tensor) + yy_channel = torch.arange(y_dim).repeat(1, x_dim, 1).transpose( + 1, 2).to(input_tensor) + + xx_channel = xx_channel / (x_dim - 1) + yy_channel = yy_channel / (y_dim - 1) + + xx_channel = xx_channel * 2 - 1 + yy_channel = yy_channel * 2 - 1 + + xx_channel = xx_channel.repeat(batch_size, 1, 1, 1).transpose(2, 3) + yy_channel = yy_channel.repeat(batch_size, 1, 1, 1).transpose(2, 3) + + ret = torch.cat( + [ # noqa + input_tensor, # noqa + xx_channel.type_as(input_tensor), # noqa + yy_channel.type_as(input_tensor) # noqa + ], # noqa + dim=1) # noqa + + if self.with_r: + rr = torch.sqrt( + torch.pow(xx_channel - 0.5, 2) + + torch.pow(yy_channel - 0.5, 2)) + ret = torch.cat([ret, rr], dim=1) + + return ret + + +class CoordConv(nn.Module): + + def __init__(self, in_channels, out_channels, with_r=False, **kwargs): + super().__init__() + self.addcoords = AddCoords(with_r=with_r) + in_channels += 2 + if with_r: + in_channels += 1 + self.conv = nn.Conv2d(in_channels, out_channels, **kwargs) + + def forward(self, x): + ret = self.addcoords(x) + ret = self.conv(ret) + return ret diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/stackedHGNetV1.py b/modelscope/models/cv/facial_68ldk_detection/lib/backbone/stackedHGNetV1.py index c81c30f8..f330cc03 100644 --- a/modelscope/models/cv/facial_68ldk_detection/lib/backbone/stackedHGNetV1.py +++ b/modelscope/models/cv/facial_68ldk_detection/lib/backbone/stackedHGNetV1.py @@ -1,307 +1,374 @@ -import numpy as np - -import torch -import torch.nn as nn -import torch.nn.functional as F - -from .core.coord_conv import CoordConvTh -from ..dataset import get_decoder - - - -class Activation(nn.Module): - def __init__(self, kind: str = 'relu', channel=None): - super().__init__() - self.kind = kind - - if '+' in kind: - norm_str, act_str = kind.split('+') - else: - norm_str, act_str = 'none', kind - - self.norm_fn = { - 'in': F.instance_norm, - 'bn': nn.BatchNorm2d(channel), - 'bn_noaffine': nn.BatchNorm2d(channel, affine=False, track_running_stats=True), - 'none': None - }[norm_str] - - self.act_fn = { - 'relu': F.relu, - 'softplus': nn.Softplus(), - 'exp': torch.exp, - 'sigmoid': torch.sigmoid, - 'tanh': torch.tanh, - 'none': None - }[act_str] - - self.channel = channel - - def forward(self, x): - if self.norm_fn is not None: - x = self.norm_fn(x) - if self.act_fn is not None: - x = self.act_fn(x) - return x - - def extra_repr(self): - return f'kind={self.kind}, channel={self.channel}' - - -class ConvBlock(nn.Module): - def __init__(self, inp_dim, out_dim, kernel_size=3, stride=1, bn=False, relu=True, groups=1): - super(ConvBlock, self).__init__() - self.inp_dim = inp_dim - self.conv = nn.Conv2d(inp_dim, out_dim, kernel_size, - stride, padding=(kernel_size - 1) // 2, groups=groups, bias=True) - self.relu = None - self.bn = None - if relu: - self.relu = nn.ReLU() - if bn: - self.bn = nn.BatchNorm2d(out_dim) - - def forward(self, x): - x = self.conv(x) - if self.bn is not None: - x = self.bn(x) - if self.relu is not None: - x = self.relu(x) - return x - - -class ResBlock(nn.Module): - def __init__(self, inp_dim, out_dim, mid_dim=None): - super(ResBlock, self).__init__() - if mid_dim is None: - mid_dim = out_dim // 2 - self.relu = nn.ReLU() - self.bn1 = nn.BatchNorm2d(inp_dim) - self.conv1 = ConvBlock(inp_dim, mid_dim, 1, relu=False) - self.bn2 = nn.BatchNorm2d(mid_dim) - self.conv2 = ConvBlock(mid_dim, mid_dim, 3, relu=False) - self.bn3 = nn.BatchNorm2d(mid_dim) - self.conv3 = ConvBlock(mid_dim, out_dim, 1, relu=False) - self.skip_layer = ConvBlock(inp_dim, out_dim, 1, relu=False) - if inp_dim == out_dim: - self.need_skip = False - else: - self.need_skip = True - - def forward(self, x): - if self.need_skip: - residual = self.skip_layer(x) - else: - residual = x - out = self.bn1(x) - out = self.relu(out) - out = self.conv1(out) - out = self.bn2(out) - out = self.relu(out) - out = self.conv2(out) - out = self.bn3(out) - out = self.relu(out) - out = self.conv3(out) - out += residual - return out - - -class Hourglass(nn.Module): - def __init__(self, n, f, increase=0, up_mode='nearest', - add_coord=False, first_one=False, x_dim=64, y_dim=64): - super(Hourglass, self).__init__() - nf = f + increase - - Block = ResBlock - - if add_coord: - self.coordconv = CoordConvTh(x_dim=x_dim, y_dim=y_dim, - with_r=True, with_boundary=True, - relu=False, bn=False, - in_channels=f, out_channels=f, - first_one=first_one, - kernel_size=1, - stride=1, padding=0) - else: - self.coordconv = None - self.up1 = Block(f, f) - - # Lower branch - self.pool1 = nn.MaxPool2d(kernel_size=2, stride=2) - - self.low1 = Block(f, nf) - self.n = n - # Recursive hourglass - if self.n > 1: - self.low2 = Hourglass(n=n - 1, f=nf, increase=increase, up_mode=up_mode, add_coord=False) - else: - self.low2 = Block(nf, nf) - self.low3 = Block(nf, f) - self.up2 = nn.Upsample(scale_factor=2, mode=up_mode) - - def forward(self, x, heatmap=None): - if self.coordconv is not None: - x = self.coordconv(x, heatmap) - up1 = self.up1(x) - pool1 = self.pool1(x) - low1 = self.low1(pool1) - low2 = self.low2(low1) - low3 = self.low3(low2) - up2 = self.up2(low3) - return up1 + up2 - - -class E2HTransform(nn.Module): - def __init__(self, edge_info, num_points, num_edges): - super().__init__() - - e2h_matrix = np.zeros([num_points, num_edges]) - for edge_id, isclosed_indices in enumerate(edge_info): - is_closed, indices = isclosed_indices - for point_id in indices: - e2h_matrix[point_id, edge_id] = 1 - e2h_matrix = torch.from_numpy(e2h_matrix).float() - - # pn x en x 1 x 1. - self.register_buffer('weight', e2h_matrix.view( - e2h_matrix.size(0), e2h_matrix.size(1), 1, 1)) - - # some keypoints are not coverred by any edges, - # in these cases, we must add a constant bias to their heatmap weights. - bias = ((e2h_matrix @ torch.ones(e2h_matrix.size(1)).to( - e2h_matrix)) < 0.5).to(e2h_matrix) - # pn x 1. - self.register_buffer('bias', bias) - - def forward(self, edgemaps): - # input: batch_size x en x hw x hh. - # output: batch_size x pn x hw x hh. - return F.conv2d(edgemaps, weight=self.weight, bias=self.bias) - - -class StackedHGNetV1(nn.Module): - def __init__(self, config, classes_num, edge_info, - nstack=4, nlevels=4, in_channel=256, increase=0, - add_coord=True, decoder_type='default'): - super(StackedHGNetV1, self).__init__() - - self.cfg = config - self.coder_type = decoder_type - self.decoder = get_decoder(decoder_type=decoder_type) - self.nstack = nstack - self.add_coord = add_coord - - self.num_heats = classes_num[0] - - if self.add_coord: - convBlock = CoordConvTh(x_dim=self.cfg.width, y_dim=self.cfg.height, - with_r=True, with_boundary=False, - relu=True, bn=True, - in_channels=3, out_channels=64, - kernel_size=7, - stride=2, padding=3) - else: - convBlock = ConvBlock(3, 64, 7, 2, bn=True, relu=True) - - pool = nn.MaxPool2d(kernel_size=2, stride=2) - - Block = ResBlock - - self.pre = nn.Sequential( - convBlock, - Block(64, 128), - pool, - Block(128, 128), - Block(128, in_channel) - ) - - self.hgs = nn.ModuleList( - [Hourglass(n=nlevels, f=in_channel, increase=increase, add_coord=self.add_coord, first_one=(_ == 0), - x_dim=int(self.cfg.width / self.nstack), y_dim=int(self.cfg.height / self.nstack)) - for _ in range(nstack)]) - - self.features = nn.ModuleList([ - nn.Sequential( - Block(in_channel, in_channel), - ConvBlock(in_channel, in_channel, 1, bn=True, relu=True) - ) for _ in range(nstack)]) - - self.out_heatmaps = nn.ModuleList( - [ConvBlock(in_channel, self.num_heats, 1, relu=False, bn=False) - for _ in range(nstack)]) - - if self.cfg.use_AAM: - self.num_edges = classes_num[1] - self.num_points = classes_num[2] - - self.e2h_transform = E2HTransform(edge_info, self.num_points, self.num_edges) - self.out_edgemaps = nn.ModuleList( - [ConvBlock(in_channel, self.num_edges, 1, relu=False, bn=False) - for _ in range(nstack)]) - self.out_pointmaps = nn.ModuleList( - [ConvBlock(in_channel, self.num_points, 1, relu=False, bn=False) - for _ in range(nstack)]) - self.merge_edgemaps = nn.ModuleList( - [ConvBlock(self.num_edges, in_channel, 1, relu=False, bn=False) - for _ in range(nstack - 1)]) - self.merge_pointmaps = nn.ModuleList( - [ConvBlock(self.num_points, in_channel, 1, relu=False, bn=False) - for _ in range(nstack - 1)]) - self.edgemap_act = Activation("sigmoid", self.num_edges) - self.pointmap_act = Activation("sigmoid", self.num_points) - - self.merge_features = nn.ModuleList( - [ConvBlock(in_channel, in_channel, 1, relu=False, bn=False) - for _ in range(nstack - 1)]) - self.merge_heatmaps = nn.ModuleList( - [ConvBlock(self.num_heats, in_channel, 1, relu=False, bn=False) - for _ in range(nstack - 1)]) - - self.nstack = nstack - - self.heatmap_act = Activation("in+relu", self.num_heats) - - self.inference = False - - def set_inference(self, inference): - self.inference = inference - - def forward(self, x): - x = self.pre(x) - - y, fusionmaps = [], [] - heatmaps = None - for i in range(self.nstack): - hg = self.hgs[i](x, heatmap=heatmaps) - feature = self.features[i](hg) - - heatmaps0 = self.out_heatmaps[i](feature) - heatmaps = self.heatmap_act(heatmaps0) - - if self.cfg.use_AAM: - pointmaps0 = self.out_pointmaps[i](feature) - pointmaps = self.pointmap_act(pointmaps0) - edgemaps0 = self.out_edgemaps[i](feature) - edgemaps = self.edgemap_act(edgemaps0) - mask = self.e2h_transform(edgemaps) * pointmaps - fusion_heatmaps = mask * heatmaps - else: - fusion_heatmaps = heatmaps - - landmarks = self.decoder.get_coords_from_heatmap(fusion_heatmaps) - - if i < self.nstack - 1: - x = x + self.merge_features[i](feature) + \ - self.merge_heatmaps[i](heatmaps) - if self.cfg.use_AAM: - x += self.merge_pointmaps[i](pointmaps) - x += self.merge_edgemaps[i](edgemaps) - - y.append(landmarks) - if self.cfg.use_AAM: - y.append(pointmaps) - y.append(edgemaps) - - fusionmaps.append(fusion_heatmaps) - - return y, fusionmaps, landmarks +import numpy as np +import torch +import torch.nn as nn +import torch.nn.functional as F + +from ..dataset import get_decoder +from .core.coord_conv import CoordConvTh + + +class Activation(nn.Module): + + def __init__(self, kind: str = 'relu', channel=None): + super().__init__() + self.kind = kind + + if '+' in kind: + norm_str, act_str = kind.split('+') + else: + norm_str, act_str = 'none', kind + + self.norm_fn = { + 'in': + F.instance_norm, + 'bn': + nn.BatchNorm2d(channel), + 'bn_noaffine': + nn.BatchNorm2d(channel, affine=False, track_running_stats=True), + 'none': + None + }[norm_str] + + self.act_fn = { + 'relu': F.relu, + 'softplus': nn.Softplus(), + 'exp': torch.exp, + 'sigmoid': torch.sigmoid, + 'tanh': torch.tanh, + 'none': None + }[act_str] + + self.channel = channel + + def forward(self, x): + if self.norm_fn is not None: + x = self.norm_fn(x) + if self.act_fn is not None: + x = self.act_fn(x) + return x + + def extra_repr(self): + return f'kind={self.kind}, channel={self.channel}' + + +class ConvBlock(nn.Module): + + def __init__(self, + inp_dim, + out_dim, + kernel_size=3, + stride=1, + bn=False, + relu=True, + groups=1): + super(ConvBlock, self).__init__() + self.inp_dim = inp_dim + self.conv = nn.Conv2d( + inp_dim, + out_dim, + kernel_size, + stride, + padding=(kernel_size - 1) // 2, + groups=groups, + bias=True) + self.relu = None + self.bn = None + if relu: + self.relu = nn.ReLU() + if bn: + self.bn = nn.BatchNorm2d(out_dim) + + def forward(self, x): + x = self.conv(x) + if self.bn is not None: + x = self.bn(x) + if self.relu is not None: + x = self.relu(x) + return x + + +class ResBlock(nn.Module): + + def __init__(self, inp_dim, out_dim, mid_dim=None): + super(ResBlock, self).__init__() + if mid_dim is None: + mid_dim = out_dim // 2 + self.relu = nn.ReLU() + self.bn1 = nn.BatchNorm2d(inp_dim) + self.conv1 = ConvBlock(inp_dim, mid_dim, 1, relu=False) + self.bn2 = nn.BatchNorm2d(mid_dim) + self.conv2 = ConvBlock(mid_dim, mid_dim, 3, relu=False) + self.bn3 = nn.BatchNorm2d(mid_dim) + self.conv3 = ConvBlock(mid_dim, out_dim, 1, relu=False) + self.skip_layer = ConvBlock(inp_dim, out_dim, 1, relu=False) + if inp_dim == out_dim: + self.need_skip = False + else: + self.need_skip = True + + def forward(self, x): + if self.need_skip: + residual = self.skip_layer(x) + else: + residual = x + out = self.bn1(x) + out = self.relu(out) + out = self.conv1(out) + out = self.bn2(out) + out = self.relu(out) + out = self.conv2(out) + out = self.bn3(out) + out = self.relu(out) + out = self.conv3(out) + out += residual + return out + + +class Hourglass(nn.Module): + + def __init__(self, + n, + f, + increase=0, + up_mode='nearest', + add_coord=False, + first_one=False, + x_dim=64, + y_dim=64): + super(Hourglass, self).__init__() + nf = f + increase + + Block = ResBlock + + if add_coord: + self.coordconv = CoordConvTh( + x_dim=x_dim, + y_dim=y_dim, + with_r=True, + with_boundary=True, + relu=False, + bn=False, + in_channels=f, + out_channels=f, + first_one=first_one, + kernel_size=1, + stride=1, + padding=0) + else: + self.coordconv = None + self.up1 = Block(f, f) + + # Lower branch + self.pool1 = nn.MaxPool2d(kernel_size=2, stride=2) + + self.low1 = Block(f, nf) + self.n = n + # Recursive hourglass + if self.n > 1: + self.low2 = Hourglass( + n=n - 1, + f=nf, + increase=increase, + up_mode=up_mode, + add_coord=False) + else: + self.low2 = Block(nf, nf) + self.low3 = Block(nf, f) + self.up2 = nn.Upsample(scale_factor=2, mode=up_mode) + + def forward(self, x, heatmap=None): + if self.coordconv is not None: + x = self.coordconv(x, heatmap) + up1 = self.up1(x) + pool1 = self.pool1(x) + low1 = self.low1(pool1) + low2 = self.low2(low1) + low3 = self.low3(low2) + up2 = self.up2(low3) + return up1 + up2 + + +class E2HTransform(nn.Module): + + def __init__(self, edge_info, num_points, num_edges): + super().__init__() + + e2h_matrix = np.zeros([num_points, num_edges]) + for edge_id, isclosed_indices in enumerate(edge_info): + is_closed, indices = isclosed_indices + for point_id in indices: + e2h_matrix[point_id, edge_id] = 1 + e2h_matrix = torch.from_numpy(e2h_matrix).float() + + # pn x en x 1 x 1. + self.register_buffer( + 'weight', + e2h_matrix.view(e2h_matrix.size(0), e2h_matrix.size(1), 1, 1)) + + # some keypoints are not coverred by any edges, + # in these cases, we must add a constant bias to their heatmap weights. + bias = ((e2h_matrix @ torch.ones(e2h_matrix.size(1)).to(e2h_matrix)) + < # noqa + 0.5).to(e2h_matrix) # noqa + # pn x 1. + self.register_buffer('bias', bias) + + def forward(self, edgemaps): + # input: batch_size x en x hw x hh. + # output: batch_size x pn x hw x hh. + return F.conv2d(edgemaps, weight=self.weight, bias=self.bias) + + +class StackedHGNetV1(nn.Module): + + def __init__(self, + config, + classes_num, + edge_info, + nstack=4, + nlevels=4, + in_channel=256, + increase=0, + add_coord=True, + decoder_type='default'): + super(StackedHGNetV1, self).__init__() + + self.cfg = config + self.coder_type = decoder_type + self.decoder = get_decoder(decoder_type=decoder_type) + self.nstack = nstack + self.add_coord = add_coord + + self.num_heats = classes_num[0] + + if self.add_coord: + convBlock = CoordConvTh( + x_dim=self.cfg.width, + y_dim=self.cfg.height, + with_r=True, + with_boundary=False, + relu=True, + bn=True, + in_channels=3, + out_channels=64, + kernel_size=7, + stride=2, + padding=3) + else: + convBlock = ConvBlock(3, 64, 7, 2, bn=True, relu=True) + + pool = nn.MaxPool2d(kernel_size=2, stride=2) + + Block = ResBlock + + self.pre = nn.Sequential(convBlock, Block(64, 128), pool, + Block(128, 128), Block(128, in_channel)) + + self.hgs = nn.ModuleList([ + Hourglass( + n=nlevels, + f=in_channel, + increase=increase, + add_coord=self.add_coord, + first_one=(_ == 0), + x_dim=int(self.cfg.width / self.nstack), + y_dim=int(self.cfg.height / self.nstack)) + for _ in range(nstack) + ]) + + self.features = nn.ModuleList([ + nn.Sequential( + Block(in_channel, in_channel), + ConvBlock(in_channel, in_channel, 1, bn=True, relu=True)) + for _ in range(nstack) + ]) + + self.out_heatmaps = nn.ModuleList([ + ConvBlock(in_channel, self.num_heats, 1, relu=False, bn=False) + for _ in range(nstack) + ]) + + if self.cfg.use_AAM: + self.num_edges = classes_num[1] + self.num_points = classes_num[2] + + self.e2h_transform = E2HTransform(edge_info, self.num_points, + self.num_edges) + self.out_edgemaps = nn.ModuleList([ + ConvBlock(in_channel, self.num_edges, 1, relu=False, bn=False) + for _ in range(nstack) + ]) + self.out_pointmaps = nn.ModuleList([ + ConvBlock( + in_channel, self.num_points, 1, relu=False, bn=False) + for _ in range(nstack) + ]) + self.merge_edgemaps = nn.ModuleList([ + ConvBlock(self.num_edges, in_channel, 1, relu=False, bn=False) + for _ in range(nstack - 1) + ]) + self.merge_pointmaps = nn.ModuleList([ + ConvBlock( + self.num_points, in_channel, 1, relu=False, bn=False) + for _ in range(nstack - 1) + ]) + self.edgemap_act = Activation('sigmoid', self.num_edges) + self.pointmap_act = Activation('sigmoid', self.num_points) + + self.merge_features = nn.ModuleList([ + ConvBlock(in_channel, in_channel, 1, relu=False, bn=False) + for _ in range(nstack - 1) + ]) + self.merge_heatmaps = nn.ModuleList([ + ConvBlock(self.num_heats, in_channel, 1, relu=False, bn=False) + for _ in range(nstack - 1) + ]) + + self.nstack = nstack + + self.heatmap_act = Activation('in+relu', self.num_heats) + + self.inference = False + + def set_inference(self, inference): + self.inference = inference + + def forward(self, x): + x = self.pre(x) + + y, fusionmaps = [], [] + heatmaps = None + for i in range(self.nstack): + hg = self.hgs[i](x, heatmap=heatmaps) + feature = self.features[i](hg) + + heatmaps0 = self.out_heatmaps[i](feature) + heatmaps = self.heatmap_act(heatmaps0) + + if self.cfg.use_AAM: + pointmaps0 = self.out_pointmaps[i](feature) + pointmaps = self.pointmap_act(pointmaps0) + edgemaps0 = self.out_edgemaps[i](feature) + edgemaps = self.edgemap_act(edgemaps0) + mask = self.e2h_transform(edgemaps) * pointmaps + fusion_heatmaps = mask * heatmaps + else: + fusion_heatmaps = heatmaps + + landmarks = self.decoder.get_coords_from_heatmap(fusion_heatmaps) + + if i < self.nstack - 1: + x = x + self.merge_features[i](feature) + \ + self.merge_heatmaps[i](heatmaps) + if self.cfg.use_AAM: + x += self.merge_pointmaps[i](pointmaps) + x += self.merge_edgemaps[i](edgemaps) + + y.append(landmarks) + if self.cfg.use_AAM: + y.append(pointmaps) + y.append(edgemaps) + + fusionmaps.append(fusion_heatmaps) + + return y, fusionmaps, landmarks diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__init__.py b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__init__.py index 7ff68531..bede64a7 100644 --- a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__init__.py +++ b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__init__.py @@ -1,10 +1,5 @@ -from .encoder import get_encoder -from .decoder import get_decoder -from .alignmentDataset import AlignmentDataset - -__all__ = [ - "Augmentation", - "AlignmentDataset", - "get_encoder", - "get_decoder" -] +from .alignmentDataset import AlignmentDataset +from .decoder import get_decoder +from .encoder import get_encoder + +__all__ = ['Augmentation', 'AlignmentDataset', 'get_encoder', 'get_decoder'] diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/__init__.cpython-312.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/__init__.cpython-312.pyc deleted file mode 100644 index 8e792a84..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/__init__.cpython-312.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/__init__.cpython-37.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/__init__.cpython-37.pyc deleted file mode 100644 index 95c1b129..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/__init__.cpython-37.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/__init__.cpython-39.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/__init__.cpython-39.pyc deleted file mode 100644 index 2de129f7..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/__init__.cpython-39.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/alignmentDataset.cpython-312.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/alignmentDataset.cpython-312.pyc deleted file mode 100644 index 1a5425ff..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/alignmentDataset.cpython-312.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/alignmentDataset.cpython-37.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/alignmentDataset.cpython-37.pyc deleted file mode 100644 index 7a433172..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/alignmentDataset.cpython-37.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/alignmentDataset.cpython-39.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/alignmentDataset.cpython-39.pyc deleted file mode 100644 index 78c95e15..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/alignmentDataset.cpython-39.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/augmentation.cpython-37.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/augmentation.cpython-37.pyc deleted file mode 100644 index 0ca68f38..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/augmentation.cpython-37.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/augmentation.cpython-39.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/augmentation.cpython-39.pyc deleted file mode 100644 index b69daa7f..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/__pycache__/augmentation.cpython-39.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/alignmentDataset.py b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/alignmentDataset.py index 8a58af2d..d0105489 100644 --- a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/alignmentDataset.py +++ b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/alignmentDataset.py @@ -1,314 +1,360 @@ -import os -import sys -import cv2 -import math -import copy -import hashlib -import imageio -import numpy as np -import pandas as pd -from scipy import interpolate -from PIL import Image, ImageEnhance, ImageFile - -import torch -import torch.nn.functional as F -from torch.utils.data import Dataset - -ImageFile.LOAD_TRUNCATED_IMAGES = True - -from .encoder import get_encoder - - -class AlignmentDataset(Dataset): - - def __init__(self, tsv_flie, image_dir="", transform=None, - width=256, height=256, channels=3, - means=(127.5, 127.5, 127.5), scale=1 / 127.5, - classes_num=None, crop_op=True, aug_prob=0.0, edge_info=None, flip_mapping=None, is_train=True, - encoder_type='default', - ): - super(AlignmentDataset, self).__init__() - self.use_AAM = True - self.encoder_type = encoder_type - self.encoder = get_encoder(height, width, encoder_type=encoder_type) - self.items = pd.read_csv(tsv_flie, sep="\t") - self.image_dir = image_dir - self.landmark_num = classes_num[0] - self.transform = transform - - self.image_width = width - self.image_height = height - self.channels = channels - assert self.image_width == self.image_height - - self.means = means - self.scale = scale - - self.aug_prob = aug_prob - self.edge_info = edge_info - self.is_train = is_train - std_lmk_5pts = np.array([ - 196.0, 226.0, - 316.0, 226.0, - 256.0, 286.0, - 220.0, 360.4, - 292.0, 360.4], np.float32) / 256.0 - 1.0 - std_lmk_5pts = np.reshape(std_lmk_5pts, (5, 2)) # [-1 1] - target_face_scale = 1.0 if crop_op else 1.25 - - self.augmentation = Augmentation( - is_train=self.is_train, - aug_prob=self.aug_prob, - image_size=self.image_width, - crop_op=crop_op, - std_lmk_5pts=std_lmk_5pts, - target_face_scale=target_face_scale, - flip_rate=0.5, - flip_mapping=flip_mapping, - random_shift_sigma=0.05, - random_rot_sigma=math.pi / 180 * 18, - random_scale_sigma=0.1, - random_gray_rate=0.2, - random_occ_rate=0.4, - random_blur_rate=0.3, - random_gamma_rate=0.2, - random_nose_fusion_rate=0.2) - - def _circle(self, img, pt, sigma=1.0, label_type='Gaussian'): - # Check that any part of the gaussian is in-bounds - tmp_size = sigma * 3 - ul = [int(pt[0] - tmp_size), int(pt[1] - tmp_size)] - br = [int(pt[0] + tmp_size + 1), int(pt[1] + tmp_size + 1)] - if (ul[0] > img.shape[1] - 1 or ul[1] > img.shape[0] - 1 or - br[0] - 1 < 0 or br[1] - 1 < 0): - # If not, just return the image as is - return img - - # Generate gaussian - size = 2 * tmp_size + 1 - x = np.arange(0, size, 1, np.float32) - y = x[:, np.newaxis] - x0 = y0 = size // 2 - # The gaussian is not normalized, we want the center value to equal 1 - if label_type == 'Gaussian': - g = np.exp(- ((x - x0) ** 2 + (y - y0) ** 2) / (2 * sigma ** 2)) - else: - g = sigma / (((x - x0) ** 2 + (y - y0) ** 2 + sigma ** 2) ** 1.5) - - # Usable gaussian range - g_x = max(0, -ul[0]), min(br[0], img.shape[1]) - ul[0] - g_y = max(0, -ul[1]), min(br[1], img.shape[0]) - ul[1] - # Image range - img_x = max(0, ul[0]), min(br[0], img.shape[1]) - img_y = max(0, ul[1]), min(br[1], img.shape[0]) - - img[img_y[0]:img_y[1], img_x[0]:img_x[1]] = 255 * g[g_y[0]:g_y[1], g_x[0]:g_x[1]] - return img - - def _polylines(self, img, lmks, is_closed, color=255, thickness=1, draw_mode=cv2.LINE_AA, - interpolate_mode=cv2.INTER_AREA, scale=4): - h, w = img.shape - img_scale = cv2.resize(img, (w * scale, h * scale), interpolation=interpolate_mode) - lmks_scale = (lmks * scale + 0.5).astype(np.int32) - cv2.polylines(img_scale, [lmks_scale], is_closed, color, thickness * scale, draw_mode) - img = cv2.resize(img_scale, (w, h), interpolation=interpolate_mode) - return img - - def _generate_edgemap(self, points, scale=0.25, thickness=1): - h, w = self.image_height, self.image_width - edgemaps = [] - for is_closed, indices in self.edge_info: - edgemap = np.zeros([h, w], dtype=np.float32) - # align_corners: False. - part = copy.deepcopy(points[np.array(indices)]) - - part = self._fit_curve(part, is_closed) - part[:, 0] = np.clip(part[:, 0], 0, w - 1) - part[:, 1] = np.clip(part[:, 1], 0, h - 1) - edgemap = self._polylines(edgemap, part, is_closed, 255, thickness) - - edgemaps.append(edgemap) - edgemaps = np.stack(edgemaps, axis=0) / 255.0 - edgemaps = torch.from_numpy(edgemaps).float().unsqueeze(0) - edgemaps = F.interpolate(edgemaps, size=(int(w * scale), int(h * scale)), mode='bilinear', - align_corners=False).squeeze() - return edgemaps - - def _fit_curve(self, lmks, is_closed=False, density=5): - try: - x = lmks[:, 0].copy() - y = lmks[:, 1].copy() - if is_closed: - x = np.append(x, x[0]) - y = np.append(y, y[0]) - tck, u = interpolate.splprep([x, y], s=0, per=is_closed, k=3) - # bins = (x.shape[0] - 1) * density + 1 - # lmk_x, lmk_y = interpolate.splev(np.linspace(0, 1, bins), f) - intervals = np.array([]) - for i in range(len(u) - 1): - intervals = np.concatenate((intervals, np.linspace(u[i], u[i + 1], density, endpoint=False))) - if not is_closed: - intervals = np.concatenate((intervals, [u[-1]])) - lmk_x, lmk_y = interpolate.splev(intervals, tck, der=0) - # der_x, der_y = interpolate.splev(intervals, tck, der=1) - curve_lmks = np.stack([lmk_x, lmk_y], axis=-1) - # curve_ders = np.stack([der_x, der_y], axis=-1) - # origin_indices = np.arange(0, curve_lmks.shape[0], density) - - return curve_lmks - except: - return lmks - - def _image_id(self, image_path): - if not os.path.exists(image_path): - image_path = os.path.join(self.image_dir, image_path) - return hashlib.md5(open(image_path, "rb").read()).hexdigest() - - def _load_image(self, image_path): - if not os.path.exists(image_path): - image_path = os.path.join(self.image_dir, image_path) - - try: - # img = cv2.imdecode(np.fromfile(image_path, dtype=np.uint8), cv2.IMREAD_COLOR)#HWC, BGR, [0-255] - img = cv2.imread(image_path, cv2.IMREAD_COLOR) # HWC, BGR, [0-255] - assert img is not None and len(img.shape) == 3 and img.shape[2] == 3 - except: - try: - img = imageio.imread(image_path) # HWC, RGB, [0-255] - img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR) # HWC, BGR, [0-255] - assert img is not None and len(img.shape) == 3 and img.shape[2] == 3 - except: - try: - gifImg = imageio.mimread(image_path) # BHWC, RGB, [0-255] - img = gifImg[0] # HWC, RGB, [0-255] - img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR) # HWC, BGR, [0-255] - assert img is not None and len(img.shape) == 3 and img.shape[2] == 3 - except: - img = None - return img - - def _compose_rotate_and_scale(self, angle, scale, shift_xy, from_center, to_center): - cosv = math.cos(angle) - sinv = math.sin(angle) - - fx, fy = from_center - tx, ty = to_center - - acos = scale * cosv - asin = scale * sinv - - a0 = acos - a1 = -asin - a2 = tx - acos * fx + asin * fy + shift_xy[0] - - b0 = asin - b1 = acos - b2 = ty - asin * fx - acos * fy + shift_xy[1] - - rot_scale_m = np.array([ - [a0, a1, a2], - [b0, b1, b2], - [0.0, 0.0, 1.0] - ], np.float32) - return rot_scale_m - - def _transformPoints2D(self, points, matrix): - """ - points (nx2), matrix (3x3) -> points (nx2) - """ - dtype = points.dtype - - # nx3 - points = np.concatenate([points, np.ones_like(points[:, [0]])], axis=1) - points = points @ np.transpose(matrix) # nx3 - points = points[:, :2] / points[:, [2, 2]] - return points.astype(dtype) - - def _transformPerspective(self, image, matrix, target_shape): - """ - image, matrix3x3 -> transformed_image - """ - return cv2.warpPerspective( - image, matrix, - dsize=(target_shape[1], target_shape[0]), - flags=cv2.INTER_LINEAR, borderValue=0) - - def _norm_points(self, points, h, w, align_corners=False): - if align_corners: - # [0, SIZE-1] -> [-1, +1] - des_points = points / torch.tensor([w - 1, h - 1]).to(points).view(1, 2) * 2 - 1 - else: - # [-0.5, SIZE-0.5] -> [-1, +1] - des_points = (points * 2 + 1) / torch.tensor([w, h]).to(points).view(1, 2) - 1 - des_points = torch.clamp(des_points, -1, 1) - return des_points - - def _denorm_points(self, points, h, w, align_corners=False): - if align_corners: - # [-1, +1] -> [0, SIZE-1] - des_points = (points + 1) / 2 * torch.tensor([w - 1, h - 1]).to(points).view(1, 1, 2) - else: - # [-1, +1] -> [-0.5, SIZE-0.5] - des_points = ((points + 1) * torch.tensor([w, h]).to(points).view(1, 1, 2) - 1) / 2 - return des_points - - def __len__(self): - return len(self.items) - - def __getitem__(self, index): - sample = dict() - - image_path = self.items.iloc[index, 0] - landmarks_5pts = self.items.iloc[index, 1] - landmarks_5pts = np.array(list(map(float, landmarks_5pts.split(","))), dtype=np.float32).reshape(5, 2) - landmarks_target = self.items.iloc[index, 2] - landmarks_target = np.array(list(map(float, landmarks_target.split(","))), dtype=np.float32).reshape( - self.landmark_num, 2) - scale = float(self.items.iloc[index, 3]) - center_w, center_h = float(self.items.iloc[index, 4]), float(self.items.iloc[index, 5]) - if len(self.items.iloc[index]) > 6: - tags = np.array(list(map(lambda x: int(float(x)), self.items.iloc[index, 6].split(",")))) - else: - tags = np.array([]) - - # image & keypoints alignment - image_path = image_path.replace('\\', '/') - # wflw testset - image_path = image_path.replace( - '//msr-facestore/Workspace/MSRA_EP_Allergan/users/yanghuan/training_data/wflw/rawImages/', '') - # trainset - image_path = image_path.replace('./rawImages/', '') - image_path = os.path.join(self.image_dir, image_path) - - # image path - sample["image_path"] = image_path - - img = self._load_image(image_path) # HWC, BGR, [0, 255] - assert img is not None - - # augmentation - # landmarks_target = [-0.5, edge-0.5] - img, landmarks_target, matrix = \ - self.augmentation.process(img, landmarks_target, landmarks_5pts, scale, center_w, center_h) - - landmarks = self._norm_points(torch.from_numpy(landmarks_target), self.image_height, self.image_width) - - sample["label"] = [landmarks, ] - - if self.use_AAM: - pointmap = self.encoder.generate_heatmap(landmarks_target) - edgemap = self._generate_edgemap(landmarks_target) - sample["label"] += [pointmap, edgemap] - - sample['matrix'] = matrix - - # image normalization - img = img.transpose(2, 0, 1).astype(np.float32) # CHW, BGR, [0, 255] - img[0, :, :] = (img[0, :, :] - self.means[0]) * self.scale - img[1, :, :] = (img[1, :, :] - self.means[1]) * self.scale - img[2, :, :] = (img[2, :, :] - self.means[2]) * self.scale - sample["data"] = torch.from_numpy(img) # CHW, BGR, [-1, 1] - - sample["tags"] = tags - - return sample +import copy +import hashlib +import math +import os +import sys + +import cv2 +import imageio +import numpy as np +import pandas as pd +import torch +import torch.nn.functional as F +from PIL import Image, ImageEnhance, ImageFile +from scipy import interpolate +from torch.utils.data import Dataset + +from .encoder import get_encoder + +ImageFile.LOAD_TRUNCATED_IMAGES = True + + +class AlignmentDataset(Dataset): + + def __init__( + self, + tsv_flie, + image_dir='', + transform=None, + width=256, + height=256, + channels=3, + means=(127.5, 127.5, 127.5), + scale=1 / 127.5, + classes_num=None, + crop_op=True, + aug_prob=0.0, + edge_info=None, + flip_mapping=None, + is_train=True, + encoder_type='default', + ): + super(AlignmentDataset, self).__init__() + self.use_AAM = True + self.encoder_type = encoder_type + self.encoder = get_encoder(height, width, encoder_type=encoder_type) + self.items = pd.read_csv(tsv_flie, sep='\t') + self.image_dir = image_dir + self.landmark_num = classes_num[0] + self.transform = transform + + self.image_width = width + self.image_height = height + self.channels = channels + assert self.image_width == self.image_height + + self.means = means + self.scale = scale + + self.aug_prob = aug_prob + self.edge_info = edge_info + self.is_train = is_train + std_lmk_5pts = np.array([ + 196.0, 226.0, 316.0, 226.0, 256.0, 286.0, 220.0, 360.4, 292.0, + 360.4 + ], np.float32) / 256.0 - 1.0 + std_lmk_5pts = np.reshape(std_lmk_5pts, (5, 2)) # [-1 1] + target_face_scale = 1.0 if crop_op else 1.25 + + self.augmentation = Augmentation( + is_train=self.is_train, + aug_prob=self.aug_prob, + image_size=self.image_width, + crop_op=crop_op, + std_lmk_5pts=std_lmk_5pts, + target_face_scale=target_face_scale, + flip_rate=0.5, + flip_mapping=flip_mapping, + random_shift_sigma=0.05, + random_rot_sigma=math.pi / 180 * 18, + random_scale_sigma=0.1, + random_gray_rate=0.2, + random_occ_rate=0.4, + random_blur_rate=0.3, + random_gamma_rate=0.2, + random_nose_fusion_rate=0.2) + + def _circle(self, img, pt, sigma=1.0, label_type='Gaussian'): + # Check that any part of the gaussian is in-bounds + tmp_size = sigma * 3 + ul = [int(pt[0] - tmp_size), int(pt[1] - tmp_size)] + br = [int(pt[0] + tmp_size + 1), int(pt[1] + tmp_size + 1)] + if (ul[0] > img.shape[1] - 1 or ul[1] > img.shape[0] - 1 + or br[0] - 1 < 0 or br[1] - 1 < 0): + # If not, just return the image as is + return img + + # Generate gaussian + size = 2 * tmp_size + 1 + x = np.arange(0, size, 1, np.float32) + y = x[:, np.newaxis] + x0 = y0 = size // 2 + # The gaussian is not normalized, we want the center value to equal 1 + if label_type == 'Gaussian': + g = np.exp(-((x - x0)**2 + (y - y0)**2) / (2 * sigma**2)) + else: + g = sigma / (((x - x0)**2 + (y - y0)**2 + sigma**2)**1.5) + + # Usable gaussian range + g_x = max(0, -ul[0]), min(br[0], img.shape[1]) - ul[0] + g_y = max(0, -ul[1]), min(br[1], img.shape[0]) - ul[1] + # Image range + img_x = max(0, ul[0]), min(br[0], img.shape[1]) + img_y = max(0, ul[1]), min(br[1], img.shape[0]) + + img[img_y[0]:img_y[1], + img_x[0]:img_x[1]] = 255 * g[g_y[0]:g_y[1], g_x[0]:g_x[1]] + return img + + def _polylines(self, + img, + lmks, + is_closed, + color=255, + thickness=1, + draw_mode=cv2.LINE_AA, + interpolate_mode=cv2.INTER_AREA, + scale=4): + h, w = img.shape + img_scale = cv2.resize( + img, (w * scale, h * scale), interpolation=interpolate_mode) + lmks_scale = (lmks * scale + 0.5).astype(np.int32) + cv2.polylines(img_scale, [lmks_scale], is_closed, color, + thickness * scale, draw_mode) + img = cv2.resize(img_scale, (w, h), interpolation=interpolate_mode) + return img + + def _generate_edgemap(self, points, scale=0.25, thickness=1): + h, w = self.image_height, self.image_width + edgemaps = [] + for is_closed, indices in self.edge_info: + edgemap = np.zeros([h, w], dtype=np.float32) + # align_corners: False. + part = copy.deepcopy(points[np.array(indices)]) + + part = self._fit_curve(part, is_closed) + part[:, 0] = np.clip(part[:, 0], 0, w - 1) + part[:, 1] = np.clip(part[:, 1], 0, h - 1) + edgemap = self._polylines(edgemap, part, is_closed, 255, thickness) + + edgemaps.append(edgemap) + edgemaps = np.stack(edgemaps, axis=0) / 255.0 + edgemaps = torch.from_numpy(edgemaps).float().unsqueeze(0) + edgemaps = F.interpolate( + edgemaps, + size=(int(w * scale), int(h * scale)), + mode='bilinear', + align_corners=False).squeeze() + return edgemaps + + def _fit_curve(self, lmks, is_closed=False, density=5): + try: + x = lmks[:, 0].copy() + y = lmks[:, 1].copy() + if is_closed: + x = np.append(x, x[0]) + y = np.append(y, y[0]) + tck, u = interpolate.splprep([x, y], s=0, per=is_closed, k=3) + # bins = (x.shape[0] - 1) * density + 1 + # lmk_x, lmk_y = interpolate.splev(np.linspace(0, 1, bins), f) + intervals = np.array([]) + for i in range(len(u) - 1): + intervals = np.concatenate( + (intervals, + np.linspace(u[i], u[i + 1], density, endpoint=False))) + if not is_closed: + intervals = np.concatenate((intervals, [u[-1]])) + lmk_x, lmk_y = interpolate.splev(intervals, tck, der=0) + # der_x, der_y = interpolate.splev(intervals, tck, der=1) + curve_lmks = np.stack([lmk_x, lmk_y], axis=-1) + # curve_ders = np.stack([der_x, der_y], axis=-1) + # origin_indices = np.arange(0, curve_lmks.shape[0], density) + + return curve_lmks + except Exception: + return lmks + + def _image_id(self, image_path): + if not os.path.exists(image_path): + image_path = os.path.join(self.image_dir, image_path) + return hashlib.md5(open(image_path, 'rb').read()).hexdigest() + + def _load_image(self, image_path): + if not os.path.exists(image_path): + image_path = os.path.join(self.image_dir, image_path) + + try: + # img = cv2.imdecode(np.fromfile(image_path, dtype=np.uint8), cv2.IMREAD_COLOR)#HWC, BGR, [0-255] + img = cv2.imread(image_path, cv2.IMREAD_COLOR) # HWC, BGR, [0-255] + assert img is not None and len( + img.shape) == 3 and img.shape[2] == 3 + except Exception: + try: + img = imageio.imread(image_path) # HWC, RGB, [0-255] + img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR) # HWC, BGR, [0-255] + assert img is not None and len( + img.shape) == 3 and img.shape[2] == 3 + except Exception: + try: + gifImg = imageio.mimread(image_path) # BHWC, RGB, [0-255] + img = gifImg[0] # HWC, RGB, [0-255] + img = cv2.cvtColor(img, + cv2.COLOR_RGB2BGR) # HWC, BGR, [0-255] + assert img is not None and len( + img.shape) == 3 and img.shape[2] == 3 + except Exception: + img = None + return img + + def _compose_rotate_and_scale(self, angle, scale, shift_xy, from_center, + to_center): + cosv = math.cos(angle) + sinv = math.sin(angle) + + fx, fy = from_center + tx, ty = to_center + + acos = scale * cosv + asin = scale * sinv + + a0 = acos + a1 = -asin + a2 = tx - acos * fx + asin * fy + shift_xy[0] + + b0 = asin + b1 = acos + b2 = ty - asin * fx - acos * fy + shift_xy[1] + + rot_scale_m = np.array([[a0, a1, a2], [b0, b1, b2], [0.0, 0.0, 1.0]], + np.float32) + return rot_scale_m + + def _transformPoints2D(self, points, matrix): + """ + points (nx2), matrix (3x3) -> points (nx2) + """ + dtype = points.dtype + + # nx3 + points = np.concatenate([points, np.ones_like(points[:, [0]])], axis=1) + points = points @ np.transpose(matrix) # nx3 + points = points[:, :2] / points[:, [2, 2]] + return points.astype(dtype) + + def _transformPerspective(self, image, matrix, target_shape): + """ + image, matrix3x3 -> transformed_image + """ + return cv2.warpPerspective( + image, + matrix, + dsize=(target_shape[1], target_shape[0]), + flags=cv2.INTER_LINEAR, + borderValue=0) + + def _norm_points(self, points, h, w, align_corners=False): + if align_corners: + # [0, SIZE-1] -> [-1, +1] + des_points = points / torch.tensor([w - 1, h - 1]).to(points).view( + 1, 2) * 2 - 1 + else: + # [-0.5, SIZE-0.5] -> [-1, +1] + des_points = (points * 2 + 1) / torch.tensor( + [w, h]).to(points).view(1, 2) - 1 + des_points = torch.clamp(des_points, -1, 1) + return des_points + + def _denorm_points(self, points, h, w, align_corners=False): + if align_corners: + # [-1, +1] -> [0, SIZE-1] + des_points = (points + 1) / 2 * torch.tensor( + [w - 1, h - 1]).to(points).view(1, 1, 2) + else: + # [-1, +1] -> [-0.5, SIZE-0.5] + des_points = ( + (points + 1) * torch.tensor([w, h]).to(points).view(1, 1, 2) + - 1) / 2 + return des_points + + def __len__(self): + return len(self.items) + + def __getitem__(self, index): + sample = dict() + + image_path = self.items.iloc[index, 0] + landmarks_5pts = self.items.iloc[index, 1] + landmarks_5pts = np.array( + list(map(float, landmarks_5pts.split(','))), + dtype=np.float32).reshape(5, 2) + landmarks_target = self.items.iloc[index, 2] + landmarks_target = np.array( + list(map(float, landmarks_target.split(','))), + dtype=np.float32).reshape(self.landmark_num, 2) + scale = float(self.items.iloc[index, 3]) + center_w, center_h = float(self.items.iloc[index, 4]), float( + self.items.iloc[index, 5]) + if len(self.items.iloc[index]) > 6: + tags = np.array( + list( + map(lambda x: int(float(x)), + self.items.iloc[index, 6].split(',')))) + else: + tags = np.array([]) + + # image & keypoints alignment + image_path = image_path.replace('\\', '/') + # wflw testset + image_path = image_path.replace( + '//msr-facestore/Workspace/MSRA_EP_Allergan/users/yanghuan/training_data/wflw/rawImages/', + '') + # trainset + image_path = image_path.replace('./rawImages/', '') + image_path = os.path.join(self.image_dir, image_path) + + # image path + sample['image_path'] = image_path + + img = self._load_image(image_path) # HWC, BGR, [0, 255] + assert img is not None + + # augmentation + # landmarks_target = [-0.5, edge-0.5] + img, landmarks_target, matrix = \ + self.augmentation.process(img, landmarks_target, landmarks_5pts, scale, center_w, center_h) + + landmarks = self._norm_points( + torch.from_numpy(landmarks_target), self.image_height, + self.image_width) + + sample['label'] = [ + landmarks, + ] + + if self.use_AAM: + pointmap = self.encoder.generate_heatmap(landmarks_target) + edgemap = self._generate_edgemap(landmarks_target) + sample['label'] += [pointmap, edgemap] + + sample['matrix'] = matrix + + # image normalization + img = img.transpose(2, 0, 1).astype(np.float32) # CHW, BGR, [0, 255] + img[0, :, :] = (img[0, :, :] - self.means[0]) * self.scale + img[1, :, :] = (img[1, :, :] - self.means[1]) * self.scale + img[2, :, :] = (img[2, :, :] - self.means[2]) * self.scale + sample['data'] = torch.from_numpy(img) # CHW, BGR, [-1, 1] + + sample['tags'] = tags + + return sample diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__init__.py b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__init__.py index c5d450d1..9acc9bcb 100644 --- a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__init__.py +++ b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__init__.py @@ -1,8 +1,9 @@ -from .decoder_default import decoder_default - -def get_decoder(decoder_type='default'): - if decoder_type == 'default': - decoder = decoder_default() - else: - raise NotImplementedError - return decoder \ No newline at end of file +from .decoder_default import decoder_default + + +def get_decoder(decoder_type='default'): + if decoder_type == 'default': + decoder = decoder_default() + else: + raise NotImplementedError + return decoder diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__pycache__/__init__.cpython-312.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__pycache__/__init__.cpython-312.pyc deleted file mode 100644 index 52acc899..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__pycache__/__init__.cpython-312.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__pycache__/__init__.cpython-37.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__pycache__/__init__.cpython-37.pyc deleted file mode 100644 index 3aa95757..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__pycache__/__init__.cpython-37.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__pycache__/__init__.cpython-39.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__pycache__/__init__.cpython-39.pyc deleted file mode 100644 index cb6ba801..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__pycache__/__init__.cpython-39.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__pycache__/decoder_default.cpython-312.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__pycache__/decoder_default.cpython-312.pyc deleted file mode 100644 index 5f9fd854..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__pycache__/decoder_default.cpython-312.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__pycache__/decoder_default.cpython-37.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__pycache__/decoder_default.cpython-37.pyc deleted file mode 100644 index 6a6204a1..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__pycache__/decoder_default.cpython-37.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__pycache__/decoder_default.cpython-39.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__pycache__/decoder_default.cpython-39.pyc deleted file mode 100644 index 21894dbb..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/__pycache__/decoder_default.cpython-39.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/decoder_default.py b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/decoder_default.py index 7b0b4edd..4e1c7c70 100644 --- a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/decoder_default.py +++ b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/decoder/decoder_default.py @@ -1,38 +1,39 @@ -import torch - - -class decoder_default: - def __init__(self, weight=1, use_weight_map=False): - self.weight = weight - self.use_weight_map = use_weight_map - - def _make_grid(self, h, w): - yy, xx = torch.meshgrid( - torch.arange(h).float() / (h - 1) * 2 - 1, - torch.arange(w).float() / (w - 1) * 2 - 1) - return yy, xx - - def get_coords_from_heatmap(self, heatmap): - """ - inputs: - - heatmap: batch x npoints x h x w - - outputs: - - coords: batch x npoints x 2 (x,y), [-1, +1] - - radius_sq: batch x npoints - """ - batch, npoints, h, w = heatmap.shape - if self.use_weight_map: - heatmap = heatmap * self.weight - - yy, xx = self._make_grid(h, w) - yy = yy.view(1, 1, h, w).to(heatmap) - xx = xx.view(1, 1, h, w).to(heatmap) - - heatmap_sum = torch.clamp(heatmap.sum([2, 3]), min=1e-6) - - yy_coord = (yy * heatmap).sum([2, 3]) / heatmap_sum # batch x npoints - xx_coord = (xx * heatmap).sum([2, 3]) / heatmap_sum # batch x npoints - coords = torch.stack([xx_coord, yy_coord], dim=-1) - - return coords +import torch + + +class decoder_default: + + def __init__(self, weight=1, use_weight_map=False): + self.weight = weight + self.use_weight_map = use_weight_map + + def _make_grid(self, h, w): + yy, xx = torch.meshgrid( + torch.arange(h).float() / (h - 1) * 2 - 1, + torch.arange(w).float() / (w - 1) * 2 - 1) + return yy, xx + + def get_coords_from_heatmap(self, heatmap): + """ + inputs: + - heatmap: batch x npoints x h x w + + outputs: + - coords: batch x npoints x 2 (x,y), [-1, +1] + - radius_sq: batch x npoints + """ + batch, npoints, h, w = heatmap.shape + if self.use_weight_map: + heatmap = heatmap * self.weight + + yy, xx = self._make_grid(h, w) + yy = yy.view(1, 1, h, w).to(heatmap) + xx = xx.view(1, 1, h, w).to(heatmap) + + heatmap_sum = torch.clamp(heatmap.sum([2, 3]), min=1e-6) + + yy_coord = (yy * heatmap).sum([2, 3]) / heatmap_sum # batch x npoints + xx_coord = (xx * heatmap).sum([2, 3]) / heatmap_sum # batch x npoints + coords = torch.stack([xx_coord, yy_coord], dim=-1) + + return coords diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__init__.py b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__init__.py index 42d0b6f9..60af5082 100644 --- a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__init__.py +++ b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__init__.py @@ -1,8 +1,13 @@ -from .encoder_default import encoder_default - -def get_encoder(image_height, image_width, scale=0.25, sigma=1.5, encoder_type='default'): - if encoder_type == 'default': - encoder = encoder_default(image_height, image_width, scale, sigma) - else: - raise NotImplementedError - return encoder +from .encoder_default import encoder_default + + +def get_encoder(image_height, + image_width, + scale=0.25, + sigma=1.5, + encoder_type='default'): + if encoder_type == 'default': + encoder = encoder_default(image_height, image_width, scale, sigma) + else: + raise NotImplementedError + return encoder diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__pycache__/__init__.cpython-312.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__pycache__/__init__.cpython-312.pyc deleted file mode 100644 index e5935921..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__pycache__/__init__.cpython-312.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__pycache__/__init__.cpython-37.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__pycache__/__init__.cpython-37.pyc deleted file mode 100644 index 285c6954..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__pycache__/__init__.cpython-37.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__pycache__/__init__.cpython-39.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__pycache__/__init__.cpython-39.pyc deleted file mode 100644 index 47c30144..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__pycache__/__init__.cpython-39.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__pycache__/encoder_default.cpython-312.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__pycache__/encoder_default.cpython-312.pyc deleted file mode 100644 index 3fc06e7f..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__pycache__/encoder_default.cpython-312.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__pycache__/encoder_default.cpython-37.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__pycache__/encoder_default.cpython-37.pyc deleted file mode 100644 index c8708528..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__pycache__/encoder_default.cpython-37.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__pycache__/encoder_default.cpython-39.pyc b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__pycache__/encoder_default.cpython-39.pyc deleted file mode 100644 index add6f977..00000000 Binary files a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/__pycache__/encoder_default.cpython-39.pyc and /dev/null differ diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/encoder_default.py b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/encoder_default.py index 92c22b13..8bff7942 100644 --- a/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/encoder_default.py +++ b/modelscope/models/cv/facial_68ldk_detection/lib/dataset/encoder/encoder_default.py @@ -1,63 +1,68 @@ -import copy -import numpy as np - -import torch -import torch.nn.functional as F - - -class encoder_default: - def __init__(self, image_height, image_width, scale=0.25, sigma=1.5): - self.image_height = image_height - self.image_width = image_width - self.scale = scale - self.sigma = sigma - - def generate_heatmap(self, points): - # points = (num_pts, 2) - h, w = self.image_height, self.image_width - pointmaps = [] - for i in range(len(points)): - pointmap = np.zeros([h, w], dtype=np.float32) - # align_corners: False. - point = copy.deepcopy(points[i]) - point[0] = max(0, min(w - 1, point[0])) - point[1] = max(0, min(h - 1, point[1])) - pointmap = self._circle(pointmap, point, sigma=self.sigma) - - pointmaps.append(pointmap) - pointmaps = np.stack(pointmaps, axis=0) / 255.0 - pointmaps = torch.from_numpy(pointmaps).float().unsqueeze(0) - pointmaps = F.interpolate(pointmaps, size=(int(w * self.scale), int(h * self.scale)), mode='bilinear', - align_corners=False).squeeze() - return pointmaps - - def _circle(self, img, pt, sigma=1.0, label_type='Gaussian'): - # Check that any part of the gaussian is in-bounds - tmp_size = sigma * 3 - ul = [int(pt[0] - tmp_size), int(pt[1] - tmp_size)] - br = [int(pt[0] + tmp_size + 1), int(pt[1] + tmp_size + 1)] - if (ul[0] > img.shape[1] - 1 or ul[1] > img.shape[0] - 1 or - br[0] - 1 < 0 or br[1] - 1 < 0): - # If not, just return the image as is - return img - - # Generate gaussian - size = 2 * tmp_size + 1 - x = np.arange(0, size, 1, np.float32) - y = x[:, np.newaxis] - x0 = y0 = size // 2 - # The gaussian is not normalized, we want the center value to equal 1 - if label_type == 'Gaussian': - g = np.exp(- ((x - x0) ** 2 + (y - y0) ** 2) / (2 * sigma ** 2)) - else: - g = sigma / (((x - x0) ** 2 + (y - y0) ** 2 + sigma ** 2) ** 1.5) - - # Usable gaussian range - g_x = max(0, -ul[0]), min(br[0], img.shape[1]) - ul[0] - g_y = max(0, -ul[1]), min(br[1], img.shape[0]) - ul[1] - # Image range - img_x = max(0, ul[0]), min(br[0], img.shape[1]) - img_y = max(0, ul[1]), min(br[1], img.shape[0]) - - img[img_y[0]:img_y[1], img_x[0]:img_x[1]] = 255 * g[g_y[0]:g_y[1], g_x[0]:g_x[1]] - return img +import copy + +import numpy as np +import torch +import torch.nn.functional as F + + +class encoder_default: + + def __init__(self, image_height, image_width, scale=0.25, sigma=1.5): + self.image_height = image_height + self.image_width = image_width + self.scale = scale + self.sigma = sigma + + def generate_heatmap(self, points): + # points = (num_pts, 2) + h, w = self.image_height, self.image_width + pointmaps = [] + for i in range(len(points)): + pointmap = np.zeros([h, w], dtype=np.float32) + # align_corners: False. + point = copy.deepcopy(points[i]) + point[0] = max(0, min(w - 1, point[0])) + point[1] = max(0, min(h - 1, point[1])) + pointmap = self._circle(pointmap, point, sigma=self.sigma) + + pointmaps.append(pointmap) + pointmaps = np.stack(pointmaps, axis=0) / 255.0 + pointmaps = torch.from_numpy(pointmaps).float().unsqueeze(0) + pointmaps = F.interpolate( + pointmaps, + size=(int(w * self.scale), int(h * self.scale)), + mode='bilinear', + align_corners=False).squeeze() + return pointmaps + + def _circle(self, img, pt, sigma=1.0, label_type='Gaussian'): + # Check that any part of the gaussian is in-bounds + tmp_size = sigma * 3 + ul = [int(pt[0] - tmp_size), int(pt[1] - tmp_size)] + br = [int(pt[0] + tmp_size + 1), int(pt[1] + tmp_size + 1)] + if (ul[0] > img.shape[1] - 1 or ul[1] > img.shape[0] - 1 + or br[0] - 1 < 0 or br[1] - 1 < 0): + # If not, just return the image as is + return img + + # Generate gaussian + size = 2 * tmp_size + 1 + x = np.arange(0, size, 1, np.float32) + y = x[:, np.newaxis] + x0 = y0 = size // 2 + # The gaussian is not normalized, we want the center value to equal 1 + if label_type == 'Gaussian': + g = np.exp(-((x - x0)**2 + (y - y0)**2) / (2 * sigma**2)) + else: + g = sigma / (((x - x0)**2 + (y - y0)**2 + sigma**2)**1.5) + + # Usable gaussian range + g_x = max(0, -ul[0]), min(br[0], img.shape[1]) - ul[0] + g_y = max(0, -ul[1]), min(br[1], img.shape[0]) - ul[1] + # Image range + img_x = max(0, ul[0]), min(br[0], img.shape[1]) + img_y = max(0, ul[1]), min(br[1], img.shape[0]) + + img[img_y[0]:img_y[1], + img_x[0]:img_x[1]] = 255 * g[g_y[0]:g_y[1], g_x[0]:g_x[1]] + return img diff --git a/modelscope/models/cv/facial_68ldk_detection/lib/utility.py b/modelscope/models/cv/facial_68ldk_detection/lib/utility.py index 825f2ebc..2e195761 100644 --- a/modelscope/models/cv/facial_68ldk_detection/lib/utility.py +++ b/modelscope/models/cv/facial_68ldk_detection/lib/utility.py @@ -1,52 +1,54 @@ -import json -import os.path as osp -import time -import torch -import numpy as np - -# private package -from ..conf import * -from .backbone import StackedHGNetV1 - - -def get_config(args): - config = None - config_name = args.config_name - if config_name == "alignment": - config = Alignment(args) - else: - assert NotImplementedError - - return config - - -def get_net(config): - net = None - if config.net == "stackedHGnet_v1": - net = StackedHGNetV1(config=config, - classes_num=config.classes_num, - edge_info=config.edge_info, - nstack=config.nstack, - add_coord=config.add_coord, - decoder_type=config.decoder_type) - else: - assert False - return net - - -def set_environment(config): - if config.device_id >= 0: - assert torch.cuda.is_available() and torch.cuda.device_count() > config.device_id - torch.cuda.empty_cache() - config.device = torch.device("cuda", config.device_id) - config.use_gpu = True - else: - config.device = torch.device("cpu") - config.use_gpu = False - - torch.set_default_dtype(torch.float32) - torch.set_default_tensor_type(torch.FloatTensor) - torch.set_flush_denormal(True) # ignore extremely small value - torch.backends.cudnn.benchmark = True # This flag allows you to enable the inbuilt cudnn auto-tuner to find the best algorithm to use for your hardware. - torch.autograd.set_detect_anomaly(True) - +import os.path as osp +import time + +import json +import numpy as np +import torch + +from ..conf import * +from .backbone import StackedHGNetV1 + + +def get_config(args): + config = None + config_name = args.config_name + if config_name == 'alignment': + config = Alignment(args) + else: + assert NotImplementedError + + return config + + +def get_net(config): + net = None + if config.net == 'stackedHGnet_v1': + net = StackedHGNetV1( + config=config, + classes_num=config.classes_num, + edge_info=config.edge_info, + nstack=config.nstack, + add_coord=config.add_coord, + decoder_type=config.decoder_type) + else: + assert False + return net + + +def set_environment(config): + if config.device_id >= 0: + assert torch.cuda.is_available( + ) and torch.cuda.device_count() > config.device_id + torch.cuda.empty_cache() + config.device = torch.device('cuda', config.device_id) + config.use_gpu = True + else: + config.device = torch.device('cpu') + config.use_gpu = False + + torch.set_default_dtype(torch.float32) + torch.set_default_tensor_type(torch.FloatTensor) + torch.set_flush_denormal(True) # ignore extremely small value + torch.backends.cudnn.benchmark = True + # This flag allows you to enable the inbuilt cudnn auto-tuner to find the best algorithm to use for your hardware. + torch.autograd.set_detect_anomaly(True) diff --git a/modelscope/models/cv/facial_68ldk_detection/star_model.py b/modelscope/models/cv/facial_68ldk_detection/star_model.py index 25996708..c0d37ef1 100644 --- a/modelscope/models/cv/facial_68ldk_detection/star_model.py +++ b/modelscope/models/cv/facial_68ldk_detection/star_model.py @@ -1,34 +1,35 @@ # Copyright (c) Alibaba, Inc. and its affiliates. import os -import numpy as np -import torch import cv2 import matplotlib.pyplot as plt +import numpy as np +import torch from modelscope.metainfo import Models from modelscope.models.base.base_torch_model import TorchModel from modelscope.models.builder import MODELS -from modelscope.preprocessors import LoadImage -from modelscope.models.cv.facial_68ldk_detection import infer +from modelscope.models.cv.facial_68ldk_detection import infer from modelscope.outputs import OutputKeys +from modelscope.preprocessors import LoadImage from modelscope.utils.constant import ModelFile, Tasks from modelscope.utils.logger import get_logger logger = get_logger() + @MODELS.register_module( Tasks.facial_68ldk_detection, module_name=Models.star_68ldk_detection) class FaceLandmarkDetection(TorchModel): def __init__(self, model_dir, *args, **kwargs): super().__init__(model_dir, *args, **kwargs) - + def forward(self, Inputs): return Inputs - + def postprocess(self, Inputs): return Inputs - + def inference(self, data): - return data \ No newline at end of file + return data diff --git a/modelscope/pipelines/cv/facial_68ldk_detection_pipeline.py b/modelscope/pipelines/cv/facial_68ldk_detection_pipeline.py index 60d0866b..5290af24 100644 --- a/modelscope/pipelines/cv/facial_68ldk_detection_pipeline.py +++ b/modelscope/pipelines/cv/facial_68ldk_detection_pipeline.py @@ -1,25 +1,24 @@ # Copyright (c) Alibaba, Inc. and its affiliates. -from typing import Any, Dict, Union - - -import numpy as np -import torch -import cv2 import argparse import os +from typing import Any, Dict, Union + +import cv2 +import numpy as np +import torch from modelscope.metainfo import Pipelines +from modelscope.models.cv.facial_68ldk_detection import infer from modelscope.outputs import OutputKeys from modelscope.pipelines.base import Input, Model, Pipeline from modelscope.pipelines.builder import PIPELINES from modelscope.preprocessors import LoadImage -from modelscope.models.cv.facial_68ldk_detection import infer -from modelscope.outputs import OutputKeys from modelscope.utils.constant import ModelFile, Tasks from modelscope.utils.logger import get_logger logger = get_logger() + @PIPELINES.register_module( Tasks.facial_68ldk_detection, module_name=Pipelines.facial_68ldk_detection) class FaceLandmarkDetectionPipeline(Pipeline): @@ -32,37 +31,38 @@ class FaceLandmarkDetectionPipeline(Pipeline): """ super().__init__(model=model, **kwargs) - parser = argparse.ArgumentParser(description="Evaluation script") + parser = argparse.ArgumentParser(description='Evaluation script') args = parser.parse_args() args.config_name = 'alignment' device_ids = list() if torch.cuda.is_available(): device_ids = [0] - else: + else: device_ids = [-1] model_path = os.path.join(model, 'pytorch_model.pkl') - - self.fld = infer.Alignment(args, model_path, dl_framework="pytorch", device_ids=device_ids) + + self.fld = infer.Alignment( + args, model_path, dl_framework='pytorch', device_ids=device_ids) logger.info('Face 2d landmark detection model, pipeline init') def preprocess(self, input: Input) -> Dict[str, Any]: print('start preprocess') - + image = LoadImage.convert_to_ndarray(input) image = cv2.resize(image, (256, 256)) data = {'image': image} print('finish preprocess') - + return data - + def forward(self, input: Dict[str, Any]) -> Dict[str, Any]: print('start infer') - + image = input['image'] if torch.cuda.is_available(): @@ -71,15 +71,16 @@ class FaceLandmarkDetectionPipeline(Pipeline): image_np = image.numpy() x1, y1, x2, y2 = 0, 0, 256, 256 - scale = max(x2 - x1, y2 - y1) / 180 + scale = max(x2 - x1, y2 - y1) / 180 center_w = (x1 + x2) / 2 center_h = (y1 + y2) / 2 - scale, center_w, center_h = float(scale), float(center_w), float(center_h) - + scale, center_w, center_h = float(scale), float(center_w), float( + center_h) + results = self.fld.analyze(image_np, scale, center_w, center_h) - + print('finish infer') - + return results def postprocess(self, inputs: Dict[str, Any]) -> Dict[str, Any]: diff --git a/tests/utils/test_hf_util.py b/tests/utils/test_hf_util.py index cf6b12ce..9d6b61bd 100644 --- a/tests/utils/test_hf_util.py +++ b/tests/utils/test_hf_util.py @@ -46,10 +46,10 @@ class HFUtilTest(unittest.TestCase): def test_transformer_patch(self): tokenizer = AutoTokenizer.from_pretrained( - 'skyline2006/llama-7b', revision='v1.0.1') + 'iic/nlp_structbert_sentiment-classification_chinese-base') self.assertIsNotNone(tokenizer) model = AutoModelForCausalLM.from_pretrained( - 'skyline2006/llama-7b', revision='v1.0.1') + 'iic/nlp_structbert_sentiment-classification_chinese-base') self.assertIsNotNone(model)