mirror of
https://github.com/modelscope/modelscope.git
synced 2026-09-01 19:49:03 +02:00
Merge pull request #609 from modelscope/merge_master_gitlab_1026
Merge master gitlab 1026
This commit is contained in:
@@ -150,7 +150,7 @@ echo -e "Building image with:\npython$python_version\npytorch$torch_version\nten
|
||||
docker_file_content=`cat docker/Dockerfile.ubuntu`
|
||||
if [ "$is_ci_test" != "True" ]; then
|
||||
echo "Building ModelScope lib, will install ModelScope lib to image"
|
||||
docker_file_content="${docker_file_content} \nRUN pip install --no-cache-dir numpy https://modelscope.oss-cn-beijing.aliyuncs.com/releases/build/modelscope-$modelscope_version-py3-none-any.whl && pip install --no-cache-dir -U transformers"
|
||||
docker_file_content="${docker_file_content} \nRUN pip install --no-cache-dir -U funasr transformers && pip install --no-cache-dir https://modelscope.oss-cn-beijing.aliyuncs.com/releases/build/modelscope-$modelscope_version-py3-none-any.whl "
|
||||
fi
|
||||
echo "$is_dsw"
|
||||
if [ "$is_dsw" == "False" ]; then
|
||||
|
||||
@@ -475,35 +475,37 @@ class HubApi:
|
||||
raise NotExistError('The model: %s has no revision : %s .' % (model_id, revision))
|
||||
logger.info('Development mode use revision: %s' % revision)
|
||||
else:
|
||||
if revision is None: # user not specified revision, use latest revision before release time
|
||||
revisions = self.list_model_revisions(
|
||||
model_id,
|
||||
cutoff_timestamp=release_timestamp,
|
||||
use_cookies=False if cookies is None else cookies)
|
||||
if len(revisions) == 0:
|
||||
logger.warning(('There is no version specified and there is no version in the model repository,'
|
||||
'use the master branch, which is fragile, please use it with caution!'))
|
||||
all_revisions = self.list_model_revisions(
|
||||
model_id,
|
||||
cutoff_timestamp=current_timestamp,
|
||||
use_cookies=False if cookies is None else cookies)
|
||||
if len(all_revisions) == 0:
|
||||
if revision is None or revision == MASTER_MODEL_BRANCH:
|
||||
revision = MASTER_MODEL_BRANCH
|
||||
else:
|
||||
# tags (revisions) returned from backend are guaranteed to be ordered by create-time
|
||||
# we shall obtain the latest revision created earlier than release version of this branch
|
||||
revision = revisions[0]
|
||||
logger.info(
|
||||
'Model revision not specified, use revision: %s'
|
||||
% revision)
|
||||
raise NotExistError('The model: %s has no revision: %s !' % (model_id, revision))
|
||||
else:
|
||||
# use user-specified revision
|
||||
revisions = self.list_model_revisions(
|
||||
model_id,
|
||||
cutoff_timestamp=current_timestamp,
|
||||
use_cookies=False if cookies is None else cookies)
|
||||
if revision not in revisions:
|
||||
if revision == MASTER_MODEL_BRANCH:
|
||||
logger.warning('Using the master branch is fragile, please use it with caution!')
|
||||
if revision is None: # user not specified revision, use latest revision before release time
|
||||
revisions = self.list_model_revisions(
|
||||
model_id,
|
||||
cutoff_timestamp=release_timestamp,
|
||||
use_cookies=False if cookies is None else cookies)
|
||||
if len(revisions) > 0:
|
||||
revision = revisions[0] # use latest revision before release time.
|
||||
else:
|
||||
raise NotExistError('The model: %s has no revision: %s !' %
|
||||
(model_id, revision))
|
||||
logger.info('Use user-specified model revision: %s' % revision)
|
||||
vl = '[%s]' % ','.join(all_revisions)
|
||||
raise NoValidRevisionError('Model revision should be specified from revisions: %s' % (vl))
|
||||
logger.warning('Model revision not specified, use revision: %s' % revision)
|
||||
else:
|
||||
# use user-specified revision
|
||||
if revision not in all_revisions:
|
||||
if revision == MASTER_MODEL_BRANCH:
|
||||
logger.warning('Using the master branch is fragile, please use it with caution!')
|
||||
else:
|
||||
vl = '[%s]' % ','.join(all_revisions)
|
||||
raise NotExistError('The model: %s has no revision: %s valid are: %s!' %
|
||||
(model_id, revision, vl))
|
||||
logger.info('Use user-specified model revision: %s' % revision)
|
||||
return revision
|
||||
|
||||
def get_model_branches_and_tags(
|
||||
|
||||
@@ -200,6 +200,7 @@ class Models(object):
|
||||
eres2net_sv = 'eres2net-sv'
|
||||
eres2net_aug_sv = 'eres2net-aug-sv'
|
||||
scl_sd = 'scl-sd'
|
||||
scl_sd_xvector = 'scl-sd-xvector'
|
||||
campplus_lre = 'cam++-lre'
|
||||
eres2net_lre = 'eres2net-lre'
|
||||
cluster_backend = 'cluster-backend'
|
||||
@@ -291,6 +292,7 @@ class Pipelines(object):
|
||||
image_denoise = 'nafnet-image-denoise'
|
||||
image_deblur = 'nafnet-image-deblur'
|
||||
image_editing = 'masactrl-image-editing'
|
||||
freeu_stable_diffusion_text2image = 'freeu-stable-diffusion-text2image'
|
||||
person_image_cartoon = 'unet-person-image-cartoon'
|
||||
ocr_detection = 'resnet18-ocr-detection'
|
||||
table_recognition = 'dla34-table-recognition'
|
||||
|
||||
@@ -8,6 +8,7 @@ import math
|
||||
import os
|
||||
from typing import Any, Dict, Union
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torch.nn.functional as F
|
||||
@@ -323,13 +324,18 @@ class SpeakerVerificationERes2Net(TorchModel):
|
||||
self.embedding_model.eval()
|
||||
|
||||
def forward(self, audio):
|
||||
assert len(audio.shape) == 2 and audio.shape[
|
||||
0] == 1, 'modelscope error: the shape of input audio to model needs to be [1, T]'
|
||||
# audio shape: [1, T]
|
||||
if isinstance(audio, np.ndarray):
|
||||
audio = torch.from_numpy(audio)
|
||||
if len(audio.shape) == 1:
|
||||
audio = audio.unsqueeze(0)
|
||||
assert len(
|
||||
audio.shape
|
||||
) == 2, 'modelscope error: the shape of input audio to model needs to be [N, T]'
|
||||
# audio shape: [N, T]
|
||||
feature = self.__extract_feature(audio)
|
||||
embedding = self.embedding_model(feature)
|
||||
|
||||
return embedding
|
||||
return embedding.detach().cpu()
|
||||
|
||||
def __extract_feature(self, audio):
|
||||
feature = Kaldi.fbank(audio, num_mel_bins=self.feature_dim)
|
||||
|
||||
@@ -8,6 +8,7 @@ import math
|
||||
import os
|
||||
from typing import Any, Dict, Union
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torch.nn.functional as F
|
||||
@@ -316,13 +317,18 @@ class SpeakerVerificationERes2Net(TorchModel):
|
||||
self.embedding_model.eval()
|
||||
|
||||
def forward(self, audio):
|
||||
assert len(audio.shape) == 2 and audio.shape[
|
||||
0] == 1, 'modelscope error: the shape of input audio to model needs to be [1, T]'
|
||||
# audio shape: [1, T]
|
||||
if isinstance(audio, np.ndarray):
|
||||
audio = torch.from_numpy(audio)
|
||||
if len(audio.shape) == 1:
|
||||
audio = audio.unsqueeze(0)
|
||||
assert len(
|
||||
audio.shape
|
||||
) == 2, 'modelscope error: the shape of input audio to model needs to be [N, T]'
|
||||
# audio shape: [N, T]
|
||||
feature = self.__extract_feature(audio)
|
||||
embedding = self.embedding_model(feature)
|
||||
|
||||
return embedding
|
||||
return embedding.detach().cpu()
|
||||
|
||||
def __extract_feature(self, audio):
|
||||
feature = Kaldi.fbank(audio, num_mel_bins=self.feature_dim)
|
||||
|
||||
303
modelscope/models/audio/sv/TDNN.py
Normal file
303
modelscope/models/audio/sv/TDNN.py
Normal file
@@ -0,0 +1,303 @@
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torch.nn.functional as F
|
||||
|
||||
|
||||
class Conv1d_O(nn.Module):
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
out_channels,
|
||||
kernel_size,
|
||||
input_shape=None,
|
||||
in_channels=None,
|
||||
stride=1,
|
||||
dilation=1,
|
||||
padding='same',
|
||||
groups=1,
|
||||
bias=True,
|
||||
padding_mode='reflect',
|
||||
skip_transpose=False,
|
||||
):
|
||||
super().__init__()
|
||||
self.kernel_size = kernel_size
|
||||
self.stride = stride
|
||||
self.dilation = dilation
|
||||
self.padding = padding
|
||||
self.padding_mode = padding_mode
|
||||
self.unsqueeze = False
|
||||
self.skip_transpose = skip_transpose
|
||||
|
||||
if input_shape is None and in_channels is None:
|
||||
raise ValueError('Must provide one of input_shape or in_channels')
|
||||
|
||||
if in_channels is None:
|
||||
in_channels = self._check_input_shape(input_shape)
|
||||
|
||||
self.conv = nn.Conv1d(
|
||||
in_channels,
|
||||
out_channels,
|
||||
self.kernel_size,
|
||||
stride=self.stride,
|
||||
dilation=self.dilation,
|
||||
padding=0,
|
||||
groups=groups,
|
||||
bias=bias,
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
"""Returns the output of the convolution.
|
||||
|
||||
Arguments
|
||||
---------
|
||||
x : torch.Tensor (batch, time, channel)
|
||||
input to convolve. 2d or 4d tensors are expected.
|
||||
"""
|
||||
|
||||
if not self.skip_transpose:
|
||||
x = x.transpose(1, -1)
|
||||
|
||||
if self.unsqueeze:
|
||||
x = x.unsqueeze(1)
|
||||
|
||||
if self.padding == 'same':
|
||||
x = self._manage_padding(x, self.kernel_size, self.dilation,
|
||||
self.stride)
|
||||
|
||||
elif self.padding == 'causal':
|
||||
num_pad = (self.kernel_size - 1) * self.dilation
|
||||
x = F.pad(x, (num_pad, 0))
|
||||
|
||||
elif self.padding == 'valid':
|
||||
pass
|
||||
|
||||
else:
|
||||
raise ValueError(
|
||||
"Padding must be 'same', 'valid' or 'causal'. Got "
|
||||
+ self.padding)
|
||||
|
||||
wx = self.conv(x)
|
||||
|
||||
if self.unsqueeze:
|
||||
wx = wx.squeeze(1)
|
||||
|
||||
if not self.skip_transpose:
|
||||
wx = wx.transpose(1, -1)
|
||||
|
||||
return wx
|
||||
|
||||
def _manage_padding(
|
||||
self,
|
||||
x,
|
||||
kernel_size: int,
|
||||
dilation: int,
|
||||
stride: int,
|
||||
):
|
||||
# Detecting input shape
|
||||
L_in = x.shape[-1]
|
||||
|
||||
# Time padding
|
||||
padding = get_padding_elem(L_in, stride, kernel_size, dilation)
|
||||
|
||||
# Applying padding
|
||||
x = F.pad(x, padding, mode=self.padding_mode)
|
||||
|
||||
return x
|
||||
|
||||
def _check_input_shape(self, shape):
|
||||
"""Checks the input shape and returns the number of input channels.
|
||||
"""
|
||||
|
||||
if len(shape) == 2:
|
||||
self.unsqueeze = True
|
||||
in_channels = 1
|
||||
elif self.skip_transpose:
|
||||
in_channels = shape[1]
|
||||
elif len(shape) == 3:
|
||||
in_channels = shape[2]
|
||||
else:
|
||||
raise ValueError('conv1d expects 2d, 3d inputs. Got '
|
||||
+ str(len(shape)))
|
||||
|
||||
# Kernel size must be odd
|
||||
if self.kernel_size % 2 == 0:
|
||||
raise ValueError(
|
||||
'The field kernel size must be an odd number. Got %s.' %
|
||||
(self.kernel_size))
|
||||
return in_channels
|
||||
|
||||
|
||||
# Skip transpose as much as possible for efficiency
|
||||
class Conv1d(Conv1d_O):
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(skip_transpose=True, *args, **kwargs)
|
||||
|
||||
|
||||
def get_padding_elem(L_in: int, stride: int, kernel_size: int, dilation: int):
|
||||
"""This function computes the number of elements to add for zero-padding.
|
||||
|
||||
Arguments
|
||||
---------
|
||||
L_in : int
|
||||
stride: int
|
||||
kernel_size : int
|
||||
dilation : int
|
||||
"""
|
||||
if stride > 1:
|
||||
n_steps = math.ceil(((L_in - kernel_size * dilation) / stride) + 1)
|
||||
L_out = stride * (n_steps - 1) + kernel_size * dilation
|
||||
padding = [kernel_size // 2, kernel_size // 2]
|
||||
|
||||
else:
|
||||
L_out = (L_in - dilation * (kernel_size - 1) - 1) // stride + 1
|
||||
|
||||
padding = [(L_in - L_out) // 2, (L_in - L_out) // 2]
|
||||
return padding
|
||||
|
||||
|
||||
class BatchNorm1d_O(nn.Module):
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
input_shape=None,
|
||||
input_size=None,
|
||||
eps=1e-05,
|
||||
momentum=0.1,
|
||||
affine=True,
|
||||
track_running_stats=True,
|
||||
combine_batch_time=False,
|
||||
skip_transpose=False,
|
||||
):
|
||||
super().__init__()
|
||||
self.combine_batch_time = combine_batch_time
|
||||
self.skip_transpose = skip_transpose
|
||||
|
||||
if input_size is None and skip_transpose:
|
||||
input_size = input_shape[1]
|
||||
elif input_size is None:
|
||||
input_size = input_shape[-1]
|
||||
|
||||
self.norm = nn.BatchNorm1d(
|
||||
input_size,
|
||||
eps=eps,
|
||||
momentum=momentum,
|
||||
affine=affine,
|
||||
track_running_stats=track_running_stats,
|
||||
)
|
||||
|
||||
def forward(self, x):
|
||||
"""Returns the normalized input tensor.
|
||||
|
||||
Arguments
|
||||
---------
|
||||
x : torch.Tensor (batch, time, [channels])
|
||||
input to normalize. 2d or 3d tensors are expected in input
|
||||
4d tensors can be used when combine_dims=True.
|
||||
"""
|
||||
shape_or = x.shape
|
||||
if self.combine_batch_time:
|
||||
if x.ndim == 3:
|
||||
x = x.reshape(shape_or[0] * shape_or[1], shape_or[2])
|
||||
else:
|
||||
x = x.reshape(shape_or[0] * shape_or[1], shape_or[3],
|
||||
shape_or[2])
|
||||
|
||||
elif not self.skip_transpose:
|
||||
x = x.transpose(-1, 1)
|
||||
|
||||
x_n = self.norm(x)
|
||||
|
||||
if self.combine_batch_time:
|
||||
x_n = x_n.reshape(shape_or)
|
||||
elif not self.skip_transpose:
|
||||
x_n = x_n.transpose(1, -1)
|
||||
|
||||
return x_n
|
||||
|
||||
|
||||
class BatchNorm1d(BatchNorm1d_O):
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(skip_transpose=True, *args, **kwargs)
|
||||
|
||||
|
||||
class Xvector(torch.nn.Module):
|
||||
"""This model extracts X-vectors for speaker recognition and diarization.
|
||||
|
||||
Arguments
|
||||
---------
|
||||
device : str
|
||||
Device used e.g. "cpu" or "cuda".
|
||||
activation : torch class
|
||||
A class for constructing the activation layers.
|
||||
tdnn_blocks : int
|
||||
Number of time-delay neural (TDNN) layers.
|
||||
tdnn_channels : list of ints
|
||||
Output channels for TDNN layer.
|
||||
tdnn_kernel_sizes : list of ints
|
||||
List of kernel sizes for each TDNN layer.
|
||||
tdnn_dilations : list of ints
|
||||
List of dilations for kernels in each TDNN layer.
|
||||
lin_neurons : int
|
||||
Number of neurons in linear layers.
|
||||
|
||||
Example
|
||||
-------
|
||||
>>> compute_xvect = Xvector('cpu')
|
||||
>>> input_feats = torch.rand([5, 10, 40])
|
||||
>>> outputs = compute_xvect(input_feats)
|
||||
>>> outputs.shape
|
||||
torch.Size([5, 1, 512])
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
device='cpu',
|
||||
activation=torch.nn.LeakyReLU,
|
||||
tdnn_blocks=5,
|
||||
tdnn_channels=[512, 512, 512, 512, 1500],
|
||||
tdnn_kernel_sizes=[5, 3, 3, 1, 1],
|
||||
tdnn_dilations=[1, 2, 3, 1, 1],
|
||||
lin_neurons=512,
|
||||
in_channels=80,
|
||||
):
|
||||
|
||||
super().__init__()
|
||||
self.blocks = nn.ModuleList()
|
||||
|
||||
# TDNN layers
|
||||
for block_index in range(tdnn_blocks):
|
||||
out_channels = tdnn_channels[block_index]
|
||||
self.blocks.extend([
|
||||
Conv1d(
|
||||
in_channels=in_channels,
|
||||
out_channels=out_channels,
|
||||
kernel_size=tdnn_kernel_sizes[block_index],
|
||||
dilation=tdnn_dilations[block_index],
|
||||
),
|
||||
activation(),
|
||||
BatchNorm1d(input_size=out_channels),
|
||||
])
|
||||
in_channels = tdnn_channels[block_index]
|
||||
|
||||
def forward(self, x, lens=None):
|
||||
"""Returns the x-vectors.
|
||||
|
||||
Arguments
|
||||
---------
|
||||
x : torch.Tensor
|
||||
"""
|
||||
|
||||
x = x.transpose(1, 2)
|
||||
|
||||
for layer in self.blocks:
|
||||
try:
|
||||
x = layer(x, lengths=lens)
|
||||
except TypeError:
|
||||
x = layer(x)
|
||||
x = x.transpose(1, 2)
|
||||
return x
|
||||
329
modelscope/models/audio/sv/speaker_change_locator_xvector.py
Normal file
329
modelscope/models/audio/sv/speaker_change_locator_xvector.py
Normal file
@@ -0,0 +1,329 @@
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
|
||||
import os
|
||||
from collections import OrderedDict
|
||||
from typing import Any, Dict, Union
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torch.nn.functional as F
|
||||
import torchaudio.compliance.kaldi as Kaldi
|
||||
|
||||
from modelscope.metainfo import Models
|
||||
from modelscope.models import MODELS, TorchModel
|
||||
from modelscope.models.audio.sv.TDNN import Xvector
|
||||
from modelscope.utils.constant import Tasks
|
||||
from modelscope.utils.device import create_device
|
||||
|
||||
|
||||
class MultiHeadSelfAttention(nn.Module):
|
||||
|
||||
def __init__(self, n_units, h=8, dropout=0.1):
|
||||
super(MultiHeadSelfAttention, self).__init__()
|
||||
self.linearQ = nn.Linear(n_units, n_units)
|
||||
self.linearK = nn.Linear(n_units, n_units)
|
||||
self.linearV = nn.Linear(n_units, n_units)
|
||||
self.linearO = nn.Linear(n_units, n_units)
|
||||
self.d_k = n_units // h
|
||||
self.h = h
|
||||
self.dropout = nn.Dropout(p=dropout)
|
||||
self.att = None
|
||||
|
||||
def forward(self, x, batch_size):
|
||||
# x: (BT, F)
|
||||
q = self.linearQ(x).reshape(batch_size, -1, self.h, self.d_k)
|
||||
k = self.linearK(x).reshape(batch_size, -1, self.h, self.d_k)
|
||||
v = self.linearV(x).reshape(batch_size, -1, self.h, self.d_k)
|
||||
scores = torch.matmul(q.transpose(1, 2), k.permute(
|
||||
0, 2, 3, 1)) / np.sqrt(self.d_k)
|
||||
# scores: (B, h, T, T)
|
||||
self.att = F.softmax(scores, dim=3)
|
||||
p_att = self.dropout(self.att)
|
||||
# v : (B, T, h, d_k)
|
||||
# p_att : (B, h, T, T)
|
||||
x = torch.matmul(p_att, v.transpose(1, 2))
|
||||
# x : (B, h, T, d_k)
|
||||
x = x.transpose(1, 2).reshape(-1, self.h * self.d_k)
|
||||
return self.linearO(x)
|
||||
|
||||
|
||||
class PositionwiseFeedForward(nn.Module):
|
||||
|
||||
def __init__(self, n_units, d_units, dropout):
|
||||
super(PositionwiseFeedForward, self).__init__()
|
||||
self.linear1 = nn.Linear(n_units, d_units)
|
||||
self.linear2 = nn.Linear(d_units, n_units)
|
||||
self.dropout = nn.Dropout(p=dropout)
|
||||
|
||||
def forward(self, x):
|
||||
return self.linear2(self.dropout(F.relu(self.linear1(x))))
|
||||
|
||||
|
||||
class PosEncoding(nn.Module):
|
||||
|
||||
def __init__(self, max_seq_len, d_word_vec):
|
||||
super(PosEncoding, self).__init__()
|
||||
pos_enc = np.array([[
|
||||
pos / np.power(10000, 2.0 * (j // 2) / d_word_vec)
|
||||
for j in range(d_word_vec)
|
||||
] for pos in range(max_seq_len)])
|
||||
pos_enc[:, 0::2] = np.sin(pos_enc[:, 0::2])
|
||||
pos_enc[:, 1::2] = np.cos(pos_enc[:, 1::2])
|
||||
pad_row = np.zeros([1, d_word_vec])
|
||||
pos_enc = np.concatenate([pad_row, pos_enc]).astype(np.float32)
|
||||
|
||||
self.pos_enc = torch.nn.Embedding(max_seq_len + 1, d_word_vec)
|
||||
self.pos_enc.weight = torch.nn.Parameter(
|
||||
torch.from_numpy(pos_enc), requires_grad=False)
|
||||
|
||||
def forward(self, input_len):
|
||||
max_len = torch.max(input_len)
|
||||
input_pos = torch.LongTensor([
|
||||
list(range(1, len + 1)) + [0] * (max_len - len)
|
||||
for len in input_len
|
||||
])
|
||||
|
||||
input_pos = input_pos.to(list(self.pos_enc.parameters())[0].device)
|
||||
return self.pos_enc(input_pos)
|
||||
|
||||
|
||||
class TransformerEncoder(nn.Module):
|
||||
|
||||
def __init__(self,
|
||||
idim,
|
||||
n_units=256,
|
||||
n_layers=2,
|
||||
e_units=512,
|
||||
h=4,
|
||||
dropout=0.1):
|
||||
super(TransformerEncoder, self).__init__()
|
||||
self.linear_in = nn.Linear(idim, n_units)
|
||||
self.lnorm_in = nn.LayerNorm(n_units)
|
||||
|
||||
self.n_layers = n_layers
|
||||
self.dropout = nn.Dropout(p=dropout)
|
||||
for i in range(n_layers):
|
||||
setattr(self, '{}{:d}'.format('lnorm1_', i), nn.LayerNorm(n_units))
|
||||
setattr(self, '{}{:d}'.format('self_att_', i),
|
||||
MultiHeadSelfAttention(n_units, h))
|
||||
setattr(self, '{}{:d}'.format('lnorm2_', i), nn.LayerNorm(n_units))
|
||||
setattr(self, '{}{:d}'.format('ff_', i),
|
||||
PositionwiseFeedForward(n_units, e_units, dropout))
|
||||
self.lnorm_out = nn.LayerNorm(n_units)
|
||||
|
||||
def forward(self, x):
|
||||
# x: [B, num_anchors, T, n_in]
|
||||
bs, num, tframe, dim = x.size()
|
||||
x = x.reshape(bs * num, tframe, -1) # [B*num_anchors, T, dim]
|
||||
# x: (B, T, F) ... batch, time, (mel)freq
|
||||
B_size, T_size, _ = x.shape
|
||||
# e: (BT, F)
|
||||
e = self.linear_in(x.reshape(B_size * T_size, -1))
|
||||
# Encoder stack
|
||||
for i in range(self.n_layers):
|
||||
# layer normalization
|
||||
e = getattr(self, '{}{:d}'.format('lnorm1_', i))(e)
|
||||
# self-attention
|
||||
s = getattr(self, '{}{:d}'.format('self_att_', i))(e, x.shape[0])
|
||||
# residual
|
||||
e = e + self.dropout(s)
|
||||
# layer normalization
|
||||
e = getattr(self, '{}{:d}'.format('lnorm2_', i))(e)
|
||||
# positionwise feed-forward
|
||||
s = getattr(self, '{}{:d}'.format('ff_', i))(e)
|
||||
# residual
|
||||
e = e + self.dropout(s)
|
||||
# final layer normalization
|
||||
# output: (BT, F)
|
||||
# output: (B, F, T)
|
||||
output = self.lnorm_out(e).reshape(B_size, T_size, -1)
|
||||
output = output.reshape(bs, num, tframe,
|
||||
-1) # [B, num_anchors, T, dim]
|
||||
return output
|
||||
|
||||
|
||||
class TransformerEncoder_out(nn.Module):
|
||||
|
||||
def __init__(self,
|
||||
idim,
|
||||
n_units=256,
|
||||
n_layers=2,
|
||||
e_units=512,
|
||||
h=4,
|
||||
dropout=0.1):
|
||||
super(TransformerEncoder_out, self).__init__()
|
||||
self.linear_in = nn.Linear(idim, n_units)
|
||||
self.lnorm_in = nn.LayerNorm(n_units)
|
||||
|
||||
self.n_layers = n_layers
|
||||
self.dropout = nn.Dropout(p=dropout)
|
||||
for i in range(n_layers):
|
||||
setattr(self, '{}{:d}'.format('lnorm1_', i), nn.LayerNorm(n_units))
|
||||
setattr(self, '{}{:d}'.format('self_att_', i),
|
||||
MultiHeadSelfAttention(n_units, h))
|
||||
setattr(self, '{}{:d}'.format('lnorm2_', i), nn.LayerNorm(n_units))
|
||||
setattr(self, '{}{:d}'.format('ff_', i),
|
||||
PositionwiseFeedForward(n_units, e_units, dropout))
|
||||
self.lnorm_out = nn.LayerNorm(n_units)
|
||||
|
||||
def forward(self, x):
|
||||
# x: (B, T, F)
|
||||
B_size, T_size, _ = x.shape
|
||||
# e: (BT, F)
|
||||
e = self.linear_in(x.reshape(B_size * T_size, -1))
|
||||
# Encoder stack
|
||||
for i in range(self.n_layers):
|
||||
# layer normalization
|
||||
e = getattr(self, '{}{:d}'.format('lnorm1_', i))(e)
|
||||
# self-attention
|
||||
s = getattr(self, '{}{:d}'.format('self_att_', i))(e, x.shape[0])
|
||||
# residual
|
||||
e = e + self.dropout(s)
|
||||
# layer normalization
|
||||
e = getattr(self, '{}{:d}'.format('lnorm2_', i))(e)
|
||||
# positionwise feed-forward
|
||||
s = getattr(self, '{}{:d}'.format('ff_', i))(e)
|
||||
# residual
|
||||
e = e + self.dropout(s)
|
||||
# final layer normalization
|
||||
# output: (BT, F)
|
||||
# output: (B, T, F)
|
||||
output = self.lnorm_out(e).reshape(B_size, T_size, -1)
|
||||
return output
|
||||
|
||||
|
||||
class OutLayer(nn.Module):
|
||||
|
||||
def __init__(self, n_units=256, num_anchors=2):
|
||||
super(OutLayer, self).__init__()
|
||||
self.rnn_combine = TransformerEncoder_out(num_anchors * n_units,
|
||||
n_units)
|
||||
self.out_linear = nn.Linear(n_units // num_anchors, 1)
|
||||
|
||||
def forward(self, input):
|
||||
# input: [B, num_anchors, T, dim]
|
||||
bs, num, tframe, dim = input.size()
|
||||
output = input.permute(0, 2, 1,
|
||||
3).reshape(bs, tframe,
|
||||
-1) # [Bs, t, num_anchors*dim]
|
||||
output = self.rnn_combine(output) # [Bs, t, n_units]
|
||||
output = output.reshape(
|
||||
bs, tframe, num, -1) # [Bs, t, num_anchors, n_units//num_anchors]
|
||||
output = self.out_linear(output).squeeze(-1) # [Bs, t, num_anchors]
|
||||
|
||||
return output
|
||||
|
||||
|
||||
class TransformerDetector(nn.Module):
|
||||
|
||||
def __init__(self,
|
||||
frame_dim=512,
|
||||
anchor_dim=192,
|
||||
hidden_dim=256,
|
||||
max_seq_len=500):
|
||||
super(TransformerDetector, self).__init__()
|
||||
self.detection = TransformerEncoder(
|
||||
idim=frame_dim + anchor_dim, n_units=hidden_dim)
|
||||
self.output = OutLayer(n_units=hidden_dim)
|
||||
self.pos_enc = PosEncoding(max_seq_len, hidden_dim)
|
||||
|
||||
def forward(self, feats, anchors):
|
||||
# feats: [1, t, fdim]
|
||||
num_frames = feats.shape[1]
|
||||
num_anchors = anchors.shape[1]
|
||||
bs = feats.shape[0]
|
||||
feats = feats.unsqueeze(1).repeat(
|
||||
1, num_anchors, 1, 1) # shape: [Bs, num_anchors, t, fdim]
|
||||
anchors = anchors.unsqueeze(2).repeat(
|
||||
1, 1, num_frames, 1) # shape: [Bs, num_anchors, t, xdim]
|
||||
sd_in = torch.cat((feats, anchors),
|
||||
dim=-1) # shape: [Bs, num_anchors, t, fdim+xdim]
|
||||
sd_out = self.detection(sd_in) # shape: [Bs, num_anchors, t, sd_dim]
|
||||
|
||||
# pos
|
||||
pos_emb = self.pos_enc(torch.tensor([num_frames] * (bs * num_anchors)))
|
||||
pos_emb = pos_emb.reshape(bs, num_anchors, num_frames, -1)
|
||||
sd_out += pos_emb
|
||||
|
||||
# output
|
||||
output = self.output(sd_out) # shape: [Bs, t, num_anchors]
|
||||
|
||||
return output
|
||||
|
||||
|
||||
@MODELS.register_module(
|
||||
Tasks.speaker_diarization, module_name=Models.scl_sd_xvector)
|
||||
class SpeakerChangeLocatorTransformer(TorchModel):
|
||||
r"""A speaekr change locator using the transformer architecture as the backbone.
|
||||
Args:
|
||||
model_dir: A model dir.
|
||||
model_config: The model config.
|
||||
"""
|
||||
|
||||
def __init__(self, model_dir, model_config: Dict[str, Any], *args,
|
||||
**kwargs):
|
||||
super().__init__(model_dir, model_config, *args, **kwargs)
|
||||
self.model_config = model_config
|
||||
|
||||
self.feature_dim = self.model_config['fbank_dim']
|
||||
frame_size = self.model_config['frame_size']
|
||||
anchor_size = self.model_config['anchor_size']
|
||||
self.device = create_device(kwargs['device'])
|
||||
|
||||
self.encoder = Xvector(in_channels=self.feature_dim)
|
||||
self.backend = TransformerDetector(
|
||||
frame_dim=frame_size, anchor_dim=anchor_size)
|
||||
|
||||
pretrained_encoder = kwargs['pretrained_encoder']
|
||||
pretrained_backend = kwargs['pretrained_backend']
|
||||
|
||||
self.__load_check_point(pretrained_encoder, pretrained_backend)
|
||||
|
||||
self.encoder.to(self.device)
|
||||
self.backend.to(self.device)
|
||||
self.encoder.eval()
|
||||
self.backend.eval()
|
||||
|
||||
def forward(self, audio, anchors):
|
||||
if isinstance(audio, np.ndarray):
|
||||
audio = torch.from_numpy(audio)
|
||||
if isinstance(anchors, np.ndarray):
|
||||
anchors = torch.from_numpy(anchors)
|
||||
assert len(audio.shape) == 2 and audio.shape[
|
||||
0] == 1, 'modelscope error: the shape of input audio to model needs to be [1, T]'
|
||||
assert len(
|
||||
anchors.shape
|
||||
) == 3 and anchors.shape[0] == 1 and anchors.shape[
|
||||
1] == 2, 'modelscope error: the shape of input anchors to model needs to be [1, 2, D]'
|
||||
# audio shape: [1, T]
|
||||
feature = self.__extract_feature(audio)
|
||||
frame_state = self.encoder(feature.to(self.device))
|
||||
output = self.backend(frame_state, anchors.to(self.device))
|
||||
output = output.squeeze(0).detach().cpu().sigmoid()
|
||||
|
||||
time_scale_factor = int(np.ceil(feature.shape[1] / output.shape[0]))
|
||||
output = output.unsqueeze(1).expand(-1, time_scale_factor,
|
||||
-1).reshape(-1, output.shape[-1])
|
||||
return output
|
||||
|
||||
def __extract_feature(self, audio):
|
||||
feature = Kaldi.fbank(audio, num_mel_bins=self.feature_dim)
|
||||
feature = feature - feature.mean(dim=0, keepdim=True)
|
||||
feature = feature.unsqueeze(0)
|
||||
return feature
|
||||
|
||||
def __load_check_point(
|
||||
self,
|
||||
pretrained_encoder,
|
||||
pretrained_backend,
|
||||
):
|
||||
self.encoder.load_state_dict(
|
||||
torch.load(
|
||||
os.path.join(self.model_dir, pretrained_encoder),
|
||||
map_location=torch.device('cpu')))
|
||||
|
||||
self.backend.load_state_dict(
|
||||
torch.load(
|
||||
os.path.join(self.model_dir, pretrained_backend),
|
||||
map_location=torch.device('cpu')))
|
||||
@@ -22,6 +22,8 @@ from modelscope.metainfo import Models
|
||||
from modelscope.models.base import Tensor
|
||||
from modelscope.models.base.base_torch_model import TorchModel
|
||||
from modelscope.models.builder import MODELS
|
||||
from modelscope.utils.compatible_with_transformers import \
|
||||
compatible_position_ids
|
||||
from modelscope.utils.config import Config
|
||||
from modelscope.utils.constant import ModelFile, Tasks
|
||||
from modelscope.utils.logger import get_logger
|
||||
@@ -88,7 +90,11 @@ class ControlNet(TorchModel):
|
||||
if device == 'gpu':
|
||||
device = 'cuda'
|
||||
model = create_model(yaml_path).cpu()
|
||||
model.load_state_dict(load_state_dict(ckpt_path, location=device))
|
||||
state_dict = load_state_dict(ckpt_path, location=device)
|
||||
compatible_position_ids(
|
||||
state_dict,
|
||||
'cond_stage_model.transformer.text_model.embeddings.position_ids')
|
||||
model.load_state_dict(state_dict)
|
||||
self.model = model.to(device)
|
||||
self.ddim_sampler = DDIMSampler(self.model)
|
||||
|
||||
|
||||
22
modelscope/models/multi_modal/freeu/__init__.py
Normal file
22
modelscope/models/multi_modal/freeu/__init__.py
Normal file
@@ -0,0 +1,22 @@
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from modelscope.utils.import_utils import LazyImportModule
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from .free_lunch_utils import register_free_upblock2d, register_free_crossattn_upblock2d
|
||||
else:
|
||||
_import_structure = {
|
||||
'free_lunch_utils':
|
||||
['register_free_upblock2d', 'register_free_crossattn_upblock2d']
|
||||
}
|
||||
|
||||
import sys
|
||||
|
||||
sys.modules[__name__] = LazyImportModule(
|
||||
__name__,
|
||||
globals()['__file__'],
|
||||
_import_structure,
|
||||
module_spec=__spec__,
|
||||
extra_objects={},
|
||||
)
|
||||
331
modelscope/models/multi_modal/freeu/free_lunch_utils.py
Normal file
331
modelscope/models/multi_modal/freeu/free_lunch_utils.py
Normal file
@@ -0,0 +1,331 @@
|
||||
# ------------------------------------------------------------------------
|
||||
# Modified from https://github.com/ChenyangSi/FreeU/blob/main/demo/free_lunch_utils.py
|
||||
# Copyright (c) 2023 TencentARC. All Rights Reserved.
|
||||
# ------------------------------------------------------------------------
|
||||
|
||||
from typing import Any, Dict, List, Optional, Tuple, Union
|
||||
|
||||
import torch
|
||||
import torch.fft as fft
|
||||
from diffusers.utils import is_torch_version
|
||||
|
||||
|
||||
def isinstance_str(x: object, cls_name: str):
|
||||
"""
|
||||
Checks whether x has any class *named* cls_name in its ancestry.
|
||||
Doesn't require access to the class's implementation.
|
||||
|
||||
Useful for patching!
|
||||
"""
|
||||
|
||||
for _cls in x.__class__.__mro__:
|
||||
if _cls.__name__ == cls_name:
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def Fourier_filter(x, threshold, scale):
|
||||
dtype = x.dtype
|
||||
x = x.type(torch.float32)
|
||||
# FFT
|
||||
x_freq = fft.fftn(x, dim=(-2, -1))
|
||||
x_freq = fft.fftshift(x_freq, dim=(-2, -1))
|
||||
|
||||
B, C, H, W = x_freq.shape
|
||||
mask = torch.ones((B, C, H, W)).cuda()
|
||||
|
||||
crow, ccol = H // 2, W // 2
|
||||
mask[..., crow - threshold:crow + threshold,
|
||||
ccol - threshold:ccol + threshold] = scale
|
||||
x_freq = x_freq * mask
|
||||
|
||||
# IFFT
|
||||
x_freq = fft.ifftshift(x_freq, dim=(-2, -1))
|
||||
x_filtered = fft.ifftn(x_freq, dim=(-2, -1)).real
|
||||
|
||||
x_filtered = x_filtered.type(dtype)
|
||||
return x_filtered
|
||||
|
||||
|
||||
def register_upblock2d(model):
|
||||
|
||||
def up_forward(self):
|
||||
|
||||
def forward(hidden_states,
|
||||
res_hidden_states_tuple,
|
||||
temb=None,
|
||||
upsample_size=None):
|
||||
for resnet in self.resnets:
|
||||
# pop res hidden states
|
||||
res_hidden_states = res_hidden_states_tuple[-1]
|
||||
res_hidden_states_tuple = res_hidden_states_tuple[:-1]
|
||||
hidden_states = torch.cat([hidden_states, res_hidden_states],
|
||||
dim=1)
|
||||
|
||||
if self.training and self.gradient_checkpointing:
|
||||
|
||||
def create_custom_forward(module):
|
||||
|
||||
def custom_forward(*inputs):
|
||||
return module(*inputs)
|
||||
|
||||
return custom_forward
|
||||
|
||||
if is_torch_version('>=', '1.11.0'):
|
||||
hidden_states = torch.utils.checkpoint.checkpoint(
|
||||
create_custom_forward(resnet),
|
||||
hidden_states,
|
||||
temb,
|
||||
use_reentrant=False)
|
||||
else:
|
||||
hidden_states = torch.utils.checkpoint.checkpoint(
|
||||
create_custom_forward(resnet), hidden_states, temb)
|
||||
else:
|
||||
hidden_states = resnet(hidden_states, temb)
|
||||
|
||||
if self.upsamplers is not None:
|
||||
for upsampler in self.upsamplers:
|
||||
hidden_states = upsampler(hidden_states, upsample_size)
|
||||
|
||||
return hidden_states
|
||||
|
||||
return forward
|
||||
|
||||
for i, upsample_block in enumerate(model.unet.up_blocks):
|
||||
if isinstance_str(upsample_block, 'UpBlock2D'):
|
||||
upsample_block.forward = up_forward(upsample_block)
|
||||
|
||||
|
||||
def register_free_upblock2d(model, b1=1.2, b2=1.4, s1=0.9, s2=0.2):
|
||||
|
||||
def up_forward(self):
|
||||
|
||||
def forward(hidden_states,
|
||||
res_hidden_states_tuple,
|
||||
temb=None,
|
||||
upsample_size=None):
|
||||
for resnet in self.resnets:
|
||||
# pop res hidden states
|
||||
res_hidden_states = res_hidden_states_tuple[-1]
|
||||
res_hidden_states_tuple = res_hidden_states_tuple[:-1]
|
||||
|
||||
# --------------- FreeU code -----------------------
|
||||
# Only operate on the first two stages
|
||||
if hidden_states.shape[1] == 1280:
|
||||
hidden_states[:, :640] = hidden_states[:, :640] * self.b1
|
||||
res_hidden_states = Fourier_filter(
|
||||
res_hidden_states, threshold=1, scale=self.s1)
|
||||
if hidden_states.shape[1] == 640:
|
||||
hidden_states[:, :320] = hidden_states[:, :320] * self.b2
|
||||
res_hidden_states = Fourier_filter(
|
||||
res_hidden_states, threshold=1, scale=self.s2)
|
||||
# ---------------------------------------------------------
|
||||
|
||||
hidden_states = torch.cat([hidden_states, res_hidden_states],
|
||||
dim=1)
|
||||
|
||||
if self.training and self.gradient_checkpointing:
|
||||
|
||||
def create_custom_forward(module):
|
||||
|
||||
def custom_forward(*inputs):
|
||||
return module(*inputs)
|
||||
|
||||
return custom_forward
|
||||
|
||||
if is_torch_version('>=', '1.11.0'):
|
||||
hidden_states = torch.utils.checkpoint.checkpoint(
|
||||
create_custom_forward(resnet),
|
||||
hidden_states,
|
||||
temb,
|
||||
use_reentrant=False)
|
||||
else:
|
||||
hidden_states = torch.utils.checkpoint.checkpoint(
|
||||
create_custom_forward(resnet), hidden_states, temb)
|
||||
else:
|
||||
hidden_states = resnet(hidden_states, temb)
|
||||
|
||||
if self.upsamplers is not None:
|
||||
for upsampler in self.upsamplers:
|
||||
hidden_states = upsampler(hidden_states, upsample_size)
|
||||
|
||||
return hidden_states
|
||||
|
||||
return forward
|
||||
|
||||
for i, upsample_block in enumerate(model.unet.up_blocks):
|
||||
if isinstance_str(upsample_block, 'UpBlock2D'):
|
||||
upsample_block.forward = up_forward(upsample_block)
|
||||
setattr(upsample_block, 'b1', b1)
|
||||
setattr(upsample_block, 'b2', b2)
|
||||
setattr(upsample_block, 's1', s1)
|
||||
setattr(upsample_block, 's2', s2)
|
||||
|
||||
|
||||
def register_crossattn_upblock2d(model):
|
||||
|
||||
def up_forward(self):
|
||||
|
||||
def forward(
|
||||
hidden_states: torch.FloatTensor,
|
||||
res_hidden_states_tuple: Tuple[torch.FloatTensor, ...],
|
||||
temb: Optional[torch.FloatTensor] = None,
|
||||
encoder_hidden_states: Optional[torch.FloatTensor] = None,
|
||||
cross_attention_kwargs: Optional[Dict[str, Any]] = None,
|
||||
upsample_size: Optional[int] = None,
|
||||
attention_mask: Optional[torch.FloatTensor] = None,
|
||||
encoder_attention_mask: Optional[torch.FloatTensor] = None,
|
||||
):
|
||||
for resnet, attn in zip(self.resnets, self.attentions):
|
||||
# pop res hidden states
|
||||
res_hidden_states = res_hidden_states_tuple[-1]
|
||||
res_hidden_states_tuple = res_hidden_states_tuple[:-1]
|
||||
hidden_states = torch.cat([hidden_states, res_hidden_states],
|
||||
dim=1)
|
||||
|
||||
if self.training and self.gradient_checkpointing:
|
||||
|
||||
def create_custom_forward(module, return_dict=None):
|
||||
|
||||
def custom_forward(*inputs):
|
||||
if return_dict is not None:
|
||||
return module(*inputs, return_dict=return_dict)
|
||||
else:
|
||||
return module(*inputs)
|
||||
|
||||
return custom_forward
|
||||
|
||||
ckpt_kwargs: Dict[str, Any] = {
|
||||
'use_reentrant': False
|
||||
} if is_torch_version('>=', '1.11.0') else {}
|
||||
hidden_states = torch.utils.checkpoint.checkpoint(
|
||||
create_custom_forward(resnet),
|
||||
hidden_states,
|
||||
temb,
|
||||
**ckpt_kwargs,
|
||||
)
|
||||
hidden_states = torch.utils.checkpoint.checkpoint(
|
||||
create_custom_forward(attn, return_dict=False),
|
||||
hidden_states,
|
||||
encoder_hidden_states,
|
||||
None, # timestep
|
||||
None, # class_labels
|
||||
cross_attention_kwargs,
|
||||
attention_mask,
|
||||
encoder_attention_mask,
|
||||
**ckpt_kwargs,
|
||||
)[0]
|
||||
else:
|
||||
hidden_states = resnet(hidden_states, temb)
|
||||
hidden_states = attn(
|
||||
hidden_states,
|
||||
encoder_hidden_states=encoder_hidden_states,
|
||||
cross_attention_kwargs=cross_attention_kwargs,
|
||||
attention_mask=attention_mask,
|
||||
encoder_attention_mask=encoder_attention_mask,
|
||||
return_dict=False,
|
||||
)[0]
|
||||
|
||||
if self.upsamplers is not None:
|
||||
for upsampler in self.upsamplers:
|
||||
hidden_states = upsampler(hidden_states, upsample_size)
|
||||
|
||||
return hidden_states
|
||||
|
||||
return forward
|
||||
|
||||
for i, upsample_block in enumerate(model.unet.up_blocks):
|
||||
if isinstance_str(upsample_block, 'CrossAttnUpBlock2D'):
|
||||
upsample_block.forward = up_forward(upsample_block)
|
||||
|
||||
|
||||
def register_free_crossattn_upblock2d(model, b1=1.2, b2=1.4, s1=0.9, s2=0.2):
|
||||
|
||||
def up_forward(self):
|
||||
|
||||
def forward(
|
||||
hidden_states: torch.FloatTensor,
|
||||
res_hidden_states_tuple: Tuple[torch.FloatTensor, ...],
|
||||
temb: Optional[torch.FloatTensor] = None,
|
||||
encoder_hidden_states: Optional[torch.FloatTensor] = None,
|
||||
cross_attention_kwargs: Optional[Dict[str, Any]] = None,
|
||||
upsample_size: Optional[int] = None,
|
||||
attention_mask: Optional[torch.FloatTensor] = None,
|
||||
encoder_attention_mask: Optional[torch.FloatTensor] = None,
|
||||
):
|
||||
for resnet, attn in zip(self.resnets, self.attentions):
|
||||
# pop res hidden states
|
||||
res_hidden_states = res_hidden_states_tuple[-1]
|
||||
res_hidden_states_tuple = res_hidden_states_tuple[:-1]
|
||||
|
||||
# --------------- FreeU code -----------------------
|
||||
# Only operate on the first two stages
|
||||
if hidden_states.shape[1] == 1280:
|
||||
hidden_states[:, :640] = hidden_states[:, :640] * self.b1
|
||||
res_hidden_states = Fourier_filter(
|
||||
res_hidden_states, threshold=1, scale=self.s1)
|
||||
if hidden_states.shape[1] == 640:
|
||||
hidden_states[:, :320] = hidden_states[:, :320] * self.b2
|
||||
res_hidden_states = Fourier_filter(
|
||||
res_hidden_states, threshold=1, scale=self.s2)
|
||||
# ---------------------------------------------------------
|
||||
|
||||
hidden_states = torch.cat([hidden_states, res_hidden_states],
|
||||
dim=1)
|
||||
|
||||
if self.training and self.gradient_checkpointing:
|
||||
|
||||
def create_custom_forward(module, return_dict=None):
|
||||
|
||||
def custom_forward(*inputs):
|
||||
if return_dict is not None:
|
||||
return module(*inputs, return_dict=return_dict)
|
||||
else:
|
||||
return module(*inputs)
|
||||
|
||||
return custom_forward
|
||||
|
||||
ckpt_kwargs: Dict[str, Any] = {
|
||||
'use_reentrant': False
|
||||
} if is_torch_version('>=', '1.11.0') else {}
|
||||
hidden_states = torch.utils.checkpoint.checkpoint(
|
||||
create_custom_forward(resnet),
|
||||
hidden_states,
|
||||
temb,
|
||||
**ckpt_kwargs,
|
||||
)
|
||||
hidden_states = torch.utils.checkpoint.checkpoint(
|
||||
create_custom_forward(attn, return_dict=False),
|
||||
hidden_states,
|
||||
encoder_hidden_states,
|
||||
None, # timestep
|
||||
None, # class_labels
|
||||
cross_attention_kwargs,
|
||||
attention_mask,
|
||||
encoder_attention_mask,
|
||||
**ckpt_kwargs,
|
||||
)[0]
|
||||
else:
|
||||
hidden_states = resnet(hidden_states, temb)
|
||||
hidden_states = attn(
|
||||
hidden_states,
|
||||
encoder_hidden_states=encoder_hidden_states,
|
||||
cross_attention_kwargs=cross_attention_kwargs,
|
||||
)[0]
|
||||
|
||||
if self.upsamplers is not None:
|
||||
for upsampler in self.upsamplers:
|
||||
hidden_states = upsampler(hidden_states, upsample_size)
|
||||
|
||||
return hidden_states
|
||||
|
||||
return forward
|
||||
|
||||
for i, upsample_block in enumerate(model.unet.up_blocks):
|
||||
if isinstance_str(upsample_block, 'CrossAttnUpBlock2D'):
|
||||
upsample_block.forward = up_forward(upsample_block)
|
||||
setattr(upsample_block, 'b1', b1)
|
||||
setattr(upsample_block, 'b2', b2)
|
||||
setattr(upsample_block, 's1', s1)
|
||||
setattr(upsample_block, 's2', s2)
|
||||
@@ -1,6 +1,7 @@
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
|
||||
import torch
|
||||
import torch.nn.functional as F
|
||||
from torch import nn
|
||||
|
||||
from modelscope.metainfo import Models
|
||||
@@ -61,8 +62,9 @@ class BertForSentenceEmbedding(BertPreTrainedModel):
|
||||
def __init__(self, config, **kwargs):
|
||||
super().__init__(config)
|
||||
self.config = config
|
||||
self.pooler_type = kwargs.get('pooler_type', 'cls')
|
||||
self.pooler_type = kwargs.get('emb_pooler_type', 'cls')
|
||||
self.pooler = Pooler(self.pooler_type)
|
||||
self.normalize = kwargs.get('normalize', False)
|
||||
setattr(self, self.base_model_prefix,
|
||||
BertModel(config, add_pooling_layer=False))
|
||||
|
||||
@@ -128,6 +130,8 @@ class BertForSentenceEmbedding(BertPreTrainedModel):
|
||||
output_hidden_states=output_hidden_states,
|
||||
return_dict=return_dict)
|
||||
outputs = self.pooler(outputs, attention_mask)
|
||||
if self.normalize:
|
||||
outputs = F.normalize(outputs, p=2, dim=-1)
|
||||
return outputs
|
||||
|
||||
@classmethod
|
||||
@@ -142,8 +146,11 @@ class BertForSentenceEmbedding(BertPreTrainedModel):
|
||||
The loaded model, which is initialized by transformers.PreTrainedModel.from_pretrained
|
||||
"""
|
||||
model_dir = kwargs.get('model_dir')
|
||||
model = super(
|
||||
Model,
|
||||
cls).from_pretrained(pretrained_model_name_or_path=model_dir)
|
||||
model_kwargs = {
|
||||
'emb_pooler_type': kwargs.get('emb_pooler_type', 'cls'),
|
||||
'normalize': kwargs.get('normalize', False)
|
||||
}
|
||||
model = super(Model, cls).from_pretrained(
|
||||
pretrained_model_name_or_path=model_dir, **model_kwargs)
|
||||
model.model_dir = model_dir
|
||||
return model
|
||||
|
||||
@@ -6,10 +6,12 @@ from modelscope.utils.import_utils import LazyImportModule
|
||||
if TYPE_CHECKING:
|
||||
from .backbone import BloomModel
|
||||
from .text_generation import BloomForTextGeneration
|
||||
from .sentence_embedding import BloomForSentenceEmbedding
|
||||
else:
|
||||
_import_structure = {
|
||||
'backbone': ['BloomModel'],
|
||||
'text_generation': ['BloomForTextGeneration'],
|
||||
'sentence_embedding': ['BloomForSentenceEmbedding']
|
||||
}
|
||||
import sys
|
||||
sys.modules[__name__] = LazyImportModule(
|
||||
|
||||
165
modelscope/models/nlp/bloom/sentence_embedding.py
Normal file
165
modelscope/models/nlp/bloom/sentence_embedding.py
Normal file
@@ -0,0 +1,165 @@
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
import torch
|
||||
from transformers import BloomConfig
|
||||
from transformers import BloomModel as BloomModelTransform
|
||||
|
||||
from modelscope.metainfo import Models
|
||||
from modelscope.models import MODELS, TorchModel
|
||||
from modelscope.outputs import SentencEmbeddingModelOutput
|
||||
from modelscope.utils.constant import Tasks
|
||||
|
||||
|
||||
class DecoderPooler(torch.nn.Module):
|
||||
"""
|
||||
Parameter-free poolers to get the sentence embedding
|
||||
'last': the last token state.
|
||||
'weighted_mean': position weighted average of all token states.
|
||||
"""
|
||||
|
||||
def __init__(self, pooler_type):
|
||||
super().__init__()
|
||||
self.pooler_type = pooler_type
|
||||
assert self.pooler_type in [
|
||||
'last', 'weighted_mean'
|
||||
], 'unrecognized pooling type %s' % self.pooler_type
|
||||
|
||||
def forward(self, outputs, attention_mask):
|
||||
last_hidden = outputs.last_hidden_state
|
||||
|
||||
if self.pooler_type in ['last']:
|
||||
n, l, h = last_hidden.shape
|
||||
|
||||
# Get shape [n] indices of the last token (i.e. the last token for each batch item)
|
||||
# Any sequence where min == 1, we use the entire sequence lenth since argmin = 0
|
||||
values, indices = torch.min(attention_mask, 1, keepdim=False)
|
||||
gather_indices = torch.where(values == 0, indices,
|
||||
l) - 1 # Shape [n]
|
||||
|
||||
# There are empty sequences, where the index would become -1 which will crash
|
||||
gather_indices = torch.clamp(gather_indices, min=0)
|
||||
|
||||
# Turn indices from shape [n] --> [n, 1, h]
|
||||
gather_indices = gather_indices.unsqueeze(1).unsqueeze(1).expand(
|
||||
n, 1, h)
|
||||
|
||||
# Gather along the 1st dim (l) (n, l, h -> n, h)
|
||||
pooled_output = torch.gather(last_hidden, 1,
|
||||
gather_indices).squeeze(dim=1)
|
||||
|
||||
elif self.pooler_type == 'weighted_mean':
|
||||
input_mask_expanded = attention_mask.unsqueeze(-1).expand(
|
||||
last_hidden.size()).float()
|
||||
# last_hidden shape: bs, seq, hidden_dim
|
||||
weights = (
|
||||
torch.arange(start=1, end=last_hidden.shape[1]
|
||||
+ 1).unsqueeze(0).unsqueeze(-1).expand(
|
||||
last_hidden.size()).float().to(
|
||||
last_hidden.device))
|
||||
assert weights.shape == last_hidden.shape == input_mask_expanded.shape
|
||||
input_mask_expanded = input_mask_expanded * weights
|
||||
|
||||
sum_embeddings = torch.sum(last_hidden * input_mask_expanded, 1)
|
||||
sum_mask = input_mask_expanded.sum(1)
|
||||
sum_mask = torch.clamp(sum_mask, min=1e-9)
|
||||
pooled_output = sum_embeddings / sum_mask
|
||||
|
||||
else:
|
||||
raise NotImplementedError
|
||||
|
||||
return pooled_output
|
||||
|
||||
|
||||
@MODELS.register_module(
|
||||
group_key=Tasks.sentence_embedding, module_name=Models.bloom)
|
||||
class BloomForSentenceEmbedding(BloomModelTransform, TorchModel):
|
||||
r"""
|
||||
This model represent a text to a dense vector by the last token state or weighted mean of all token states.
|
||||
See `Language Models are Universal Embedders
|
||||
<https://arxiv.org/pdf/2310.08232.pdf>`_ for details.
|
||||
"""
|
||||
|
||||
def __init__(self, config, **kwargs):
|
||||
super().__init__(config)
|
||||
self.config = config
|
||||
self.pooler_type = kwargs.get('emb_pooler_type', 'weighted_mean')
|
||||
self.pooler = DecoderPooler(self.pooler_type)
|
||||
self.normalize = kwargs.get('normalize', False)
|
||||
setattr(self, self.base_model_prefix, BloomModelTransform(config))
|
||||
|
||||
def forward(self, query=None, docs=None, labels=None):
|
||||
r"""
|
||||
Args:
|
||||
query (:obj: `dict`): Dict of pretrained models's input for the query sequence. See
|
||||
:meth:`transformers.PreTrainedTokenizer.encode` and :meth:`transformers.PreTrainedTokenizer.__call__`
|
||||
for details.
|
||||
docs (:obj: `dict`): Dict of pretrained models's input for the query sequence. See
|
||||
:meth:`transformers.PreTrainedTokenizer.encode` and :meth:`transformers.PreTrainedTokenizer.__call__`
|
||||
for details.
|
||||
Returns:
|
||||
Returns `modelscope.outputs.SentencEmbeddingModelOutput
|
||||
Examples:
|
||||
>>> from modelscope.models import Model
|
||||
>>> from modelscope.preprocessors import Preprocessor
|
||||
>>> model = Model.from_pretrained('damo/nlp_udever_bloom_560m')
|
||||
>>> preprocessor = Preprocessor.from_pretrained('damo/nlp_udever_bloom_560m')
|
||||
>>> inputs = preprocessor({'source_sentence': ['This is a test']})
|
||||
>>> outputs = model(**inputs)
|
||||
>>> print(outputs)
|
||||
"""
|
||||
query_embeddings, doc_embeddings = None, None
|
||||
if query is not None:
|
||||
query_embeddings = self.encode(**query)
|
||||
if docs is not None:
|
||||
doc_embeddings = self.encode(**docs)
|
||||
outputs = SentencEmbeddingModelOutput(
|
||||
query_embeddings=query_embeddings, doc_embeddings=doc_embeddings)
|
||||
if query_embeddings is None or doc_embeddings is None:
|
||||
return outputs
|
||||
if self.base_model.training:
|
||||
loss_fct = torch.nn.CrossEntropyLoss()
|
||||
scores = torch.matmul(query_embeddings, doc_embeddings.T)
|
||||
if labels is None:
|
||||
labels = torch.arange(
|
||||
scores.size(0), device=scores.device, dtype=torch.long)
|
||||
labels = labels * (
|
||||
doc_embeddings.size(0) // query_embeddings.size(0))
|
||||
loss = loss_fct(scores, labels)
|
||||
outputs.loss = loss
|
||||
return outputs
|
||||
|
||||
def encode(
|
||||
self,
|
||||
input_ids=None,
|
||||
attention_mask=None,
|
||||
):
|
||||
outputs = self.base_model.forward(
|
||||
input_ids, attention_mask=attention_mask)
|
||||
embeddings = self.pooler(outputs, attention_mask)
|
||||
if self.normalize:
|
||||
embeddings = torch.nn.functional.normalize(embeddings, p=2, dim=-1)
|
||||
return embeddings
|
||||
|
||||
@classmethod
|
||||
def _instantiate(cls, **kwargs):
|
||||
"""Instantiate the model.
|
||||
|
||||
Args:
|
||||
kwargs: Input args.
|
||||
model_dir: The model dir used to load the checkpoint and the label information.
|
||||
|
||||
Returns:
|
||||
The loaded model, which is initialized by transformers.PreTrainedModel.from_pretrained
|
||||
"""
|
||||
model_dir = kwargs.get('model_dir')
|
||||
model_kwargs = {
|
||||
'emb_pooler_type': kwargs.get('emb_pooler_type', 'weighted_mean'),
|
||||
'normalize': kwargs.get('normalize', False)
|
||||
}
|
||||
if model_dir is None:
|
||||
config = BloomConfig(**kwargs)
|
||||
model = cls(config)
|
||||
else:
|
||||
model = super(BloomModelTransform, cls).from_pretrained(
|
||||
pretrained_model_name_or_path=model_dir, **model_kwargs)
|
||||
model.model_dir = model_dir
|
||||
return model
|
||||
@@ -6,6 +6,8 @@ import torch
|
||||
from modelscope.metainfo import Models
|
||||
from modelscope.models.base import Tensor, TorchModel
|
||||
from modelscope.models.builder import MODELS
|
||||
from modelscope.utils.compatible_with_transformers import \
|
||||
compatible_position_ids
|
||||
from modelscope.utils.config import Config
|
||||
from modelscope.utils.constant import ModelFile, Tasks
|
||||
from .backbone import Re2GModel
|
||||
@@ -24,6 +26,8 @@ class DocumentGroundedDialogGenerateModel(TorchModel):
|
||||
state_dict = torch.load(
|
||||
os.path.join(self.model_dir, ModelFile.TORCH_MODEL_BIN_FILE),
|
||||
map_location='cpu')
|
||||
compatible_position_ids(
|
||||
state_dict, 'rerank.encoder.roberta.embeddings.position_ids')
|
||||
self.model.load_state_dict(state_dict)
|
||||
|
||||
def forward(self, input: Dict[str, Tensor]):
|
||||
|
||||
@@ -6,6 +6,8 @@ import torch
|
||||
from modelscope.metainfo import Models
|
||||
from modelscope.models.base import Tensor, TorchModel
|
||||
from modelscope.models.builder import MODELS
|
||||
from modelscope.utils.compatible_with_transformers import \
|
||||
compatible_position_ids
|
||||
from modelscope.utils.config import Config
|
||||
from modelscope.utils.constant import ModelFile, Tasks
|
||||
from .backbone import DPRModel
|
||||
@@ -24,6 +26,8 @@ class DocumentGroundedDialogRetrievalModel(TorchModel):
|
||||
state_dict = torch.load(
|
||||
os.path.join(self.model_dir, ModelFile.TORCH_MODEL_BIN_FILE),
|
||||
map_location='cpu')
|
||||
compatible_position_ids(state_dict,
|
||||
'ctx_encoder.encoder.embeddings.position_ids')
|
||||
self.model.load_state_dict(state_dict)
|
||||
|
||||
def forward(self, input: Dict[str, Tensor], gck_segment=32):
|
||||
|
||||
@@ -16,6 +16,8 @@ from modelscope.models.base import TorchModel
|
||||
from modelscope.models.builder import MODELS
|
||||
from modelscope.models.nlp.task_models.task_model import EncoderModel
|
||||
from modelscope.outputs import MachineReadingComprehensionOutput, OutputKeys
|
||||
from modelscope.utils.compatible_with_transformers import \
|
||||
compatible_position_ids
|
||||
from modelscope.utils.constant import ModelFile, Tasks
|
||||
from modelscope.utils.hub import parse_label_mapping
|
||||
|
||||
@@ -45,9 +47,10 @@ class ModelForMachineReadingComprehension(TorchModel):
|
||||
self.config.hidden_dropout_prob,
|
||||
intermediate_hidden_size=self.config.
|
||||
projection_intermediate_hidden_size)
|
||||
self.load_state_dict(
|
||||
torch.load(
|
||||
os.path.join(model_dir, ModelFile.TORCH_MODEL_BIN_FILE)))
|
||||
state_dict = torch.load(
|
||||
os.path.join(model_dir, ModelFile.TORCH_MODEL_BIN_FILE))
|
||||
compatible_position_ids(state_dict, 'roberta.embeddings.position_ids')
|
||||
self.load_state_dict(state_dict)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
|
||||
@@ -442,8 +442,10 @@ TASK_OUTPUTS = {
|
||||
Tasks.table_recognition: [OutputKeys.POLYGONS],
|
||||
Tasks.lineless_table_recognition: [OutputKeys.POLYGONS, OutputKeys.BOXES],
|
||||
Tasks.license_plate_detection: [OutputKeys.POLYGONS, OutputKeys.TEXT],
|
||||
Tasks.card_detection_correction:
|
||||
[OutputKeys.POLYGONS, OutputKeys.OUTPUT_IMGS],
|
||||
Tasks.card_detection_correction: [
|
||||
OutputKeys.POLYGONS, OutputKeys.SCORES, OutputKeys.OUTPUT_IMGS,
|
||||
OutputKeys.LABELS, OutputKeys.LAYOUT
|
||||
],
|
||||
|
||||
# ocr recognition result for single sample
|
||||
# {
|
||||
@@ -672,9 +674,8 @@ TASK_OUTPUTS = {
|
||||
# np.array # 2D array containing only 0, 1
|
||||
# ]
|
||||
# }
|
||||
Tasks.image_segmentation: [
|
||||
OutputKeys.SCORES, OutputKeys.LABELS, OutputKeys.MASKS
|
||||
],
|
||||
Tasks.image_segmentation:
|
||||
[OutputKeys.SCORES, OutputKeys.LABELS, OutputKeys.MASKS],
|
||||
|
||||
# video panoptic segmentation result for single sample
|
||||
# "scores": [[0.8, 0.25, 0.05, 0.05], [0.9, 0.1, 0.05, 0.05]]
|
||||
|
||||
@@ -92,8 +92,9 @@ class SegmentationClusteringPipeline(Pipeline):
|
||||
def forward(self, input: list) -> np.ndarray:
|
||||
embeddings = []
|
||||
for s in input:
|
||||
_, embs = self.sv_pipeline([s[2]], output_emb=True)
|
||||
embeddings.append(embs)
|
||||
save_dict = self.sv_pipeline([s[2]], output_emb=True)
|
||||
if save_dict['embs'].shape == (1, 192):
|
||||
embeddings.append(save_dict['embs'])
|
||||
embeddings = np.concatenate(embeddings)
|
||||
return embeddings
|
||||
|
||||
|
||||
@@ -3,8 +3,10 @@
|
||||
import io
|
||||
from typing import Any, Dict, List, Union
|
||||
|
||||
import numpy as np
|
||||
import soundfile as sf
|
||||
import torch
|
||||
import torchaudio
|
||||
|
||||
from modelscope.fileio import File
|
||||
from modelscope.metainfo import Pipelines
|
||||
@@ -46,64 +48,111 @@ class ERes2Net_Pipeline(Pipeline):
|
||||
self.model_config = self.model.model_config
|
||||
self.config = self.model.other_config
|
||||
self.thr = self.config['yesOrno_thr']
|
||||
self.save_dict = {}
|
||||
|
||||
def __call__(self,
|
||||
in_audios: List[str],
|
||||
thr: float = None) -> Dict[str, Any]:
|
||||
in_audios: Union[np.ndarray, list],
|
||||
save_dir: str = None,
|
||||
output_emb: bool = False,
|
||||
thr: float = None):
|
||||
if thr is not None:
|
||||
self.thr = thr
|
||||
if self.thr < -1 or self.thr > 1:
|
||||
raise ValueError(
|
||||
'modelscope error: the thr value should be in [-1, 1], but found to be %f.'
|
||||
% self.thr)
|
||||
outputs = self.preprocess(in_audios)
|
||||
outputs = self.forward(outputs)
|
||||
outputs = self.postprocess(outputs)
|
||||
|
||||
return outputs
|
||||
|
||||
def forward(self, inputs: Dict[str, Any]) -> Dict[str, Any]:
|
||||
emb1 = self.model(inputs['data1'])
|
||||
emb2 = self.model(inputs['data2'])
|
||||
|
||||
return {'emb1': emb1, 'emb2': emb2}
|
||||
|
||||
def postprocess(self, inputs: Dict[str, Any]) -> Dict[str, Any]:
|
||||
score = self.compute_cos_similarity(inputs['emb1'], inputs['emb2'])
|
||||
score = round(score, 5)
|
||||
if score >= self.thr:
|
||||
ans = 'yes'
|
||||
wavs = self.preprocess(in_audios)
|
||||
embs = self.forward(wavs)
|
||||
outputs = self.postprocess(embs, in_audios, save_dir)
|
||||
if output_emb:
|
||||
self.save_dict['outputs'] = outputs
|
||||
self.save_dict['embs'] = embs.numpy()
|
||||
return self.save_dict
|
||||
else:
|
||||
ans = 'no'
|
||||
return outputs
|
||||
|
||||
return {OutputKeys.SCORE: score, OutputKeys.TEXT: ans}
|
||||
def forward(self, inputs: list):
|
||||
embs = []
|
||||
for x in inputs:
|
||||
embs.append(self.model(x))
|
||||
embs = torch.cat(embs)
|
||||
return embs
|
||||
|
||||
def preprocess(self, inputs: List[str],
|
||||
**preprocess_params) -> Dict[str, Any]:
|
||||
if len(inputs) != 2:
|
||||
raise ValueError(
|
||||
'modelscope error: Two input audio files are required.')
|
||||
output = {}
|
||||
def postprocess(self,
|
||||
inputs: torch.Tensor,
|
||||
in_audios: Union[np.ndarray, list],
|
||||
save_dir=None):
|
||||
if isinstance(in_audios[0], str) and save_dir is not None:
|
||||
# save the embeddings
|
||||
os.makedirs(save_dir, exist_ok=True)
|
||||
for i, p in enumerate(in_audios):
|
||||
save_path = os.path.join(
|
||||
save_dir, '%s.npy' %
|
||||
(os.path.basename(p).rsplit('.', 1)[0]))
|
||||
np.save(save_path, inputs[i].numpy())
|
||||
|
||||
if len(inputs) == 2:
|
||||
# compute the score
|
||||
score = self.compute_cos_similarity(inputs[0], inputs[1])
|
||||
score = round(score, 5)
|
||||
if score >= self.thr:
|
||||
ans = 'yes'
|
||||
else:
|
||||
ans = 'no'
|
||||
output = {OutputKeys.SCORE: score, OutputKeys.TEXT: ans}
|
||||
else:
|
||||
output = {OutputKeys.TEXT: 'No similarity score output'}
|
||||
|
||||
return output
|
||||
|
||||
def preprocess(self, inputs: Union[np.ndarray, list]):
|
||||
output = []
|
||||
for i in range(len(inputs)):
|
||||
if isinstance(inputs[i], str):
|
||||
file_bytes = File.read(inputs[i])
|
||||
data, fs = sf.read(io.BytesIO(file_bytes), dtype='float32')
|
||||
if len(data.shape) == 2:
|
||||
data = data[:, 0]
|
||||
data = torch.from_numpy(data).unsqueeze(0)
|
||||
if fs != self.model_config['sample_rate']:
|
||||
raise ValueError(
|
||||
'modelscope error: Only support %d sample rate files'
|
||||
% self.model_cfg['sample_rate'])
|
||||
output['data%d' %
|
||||
(i + 1)] = torch.from_numpy(data).unsqueeze(0)
|
||||
logger.warning(
|
||||
'The sample rate of audio is not %d, resample it.'
|
||||
% self.model_config['sample_rate'])
|
||||
data, fs = torchaudio.sox_effects.apply_effects_tensor(
|
||||
data,
|
||||
fs,
|
||||
effects=[[
|
||||
'rate',
|
||||
str(self.model_config['sample_rate'])
|
||||
]])
|
||||
data = data.squeeze(0)
|
||||
elif isinstance(inputs[i], np.ndarray):
|
||||
assert len(
|
||||
inputs[i].shape
|
||||
) == 1, 'modelscope error: Input array should be [N, T]'
|
||||
data = inputs[i]
|
||||
if data.dtype in ['int16', 'int32', 'int64']:
|
||||
data = (data / (1 << 15)).astype('float32')
|
||||
else:
|
||||
data = data.astype('float32')
|
||||
data = torch.from_numpy(data)
|
||||
else:
|
||||
raise ValueError(
|
||||
'modelscope error: The input type is temporarily restricted to audio file address'
|
||||
% i)
|
||||
'modelscope error: The input type is restricted to audio address and nump array.'
|
||||
)
|
||||
output.append(data)
|
||||
return output
|
||||
|
||||
def compute_cos_similarity(self, emb1: torch.Tensor,
|
||||
emb2: torch.Tensor) -> float:
|
||||
def compute_cos_similarity(self, emb1: Union[np.ndarray, torch.Tensor],
|
||||
emb2: Union[np.ndarray, torch.Tensor]) -> float:
|
||||
if isinstance(emb1, np.ndarray):
|
||||
emb1 = torch.from_numpy(emb1)
|
||||
if isinstance(emb2, np.ndarray):
|
||||
emb2 = torch.from_numpy(emb2)
|
||||
if len(emb1.shape):
|
||||
emb1 = emb1.unsqueeze(0)
|
||||
if len(emb2.shape):
|
||||
emb2 = emb2.unsqueeze(0)
|
||||
assert len(emb1.shape) == 2 and len(emb2.shape) == 2
|
||||
cos = torch.nn.CosineSimilarity(dim=1, eps=1e-6)
|
||||
cosine = cos(emb1, emb2)
|
||||
|
||||
@@ -50,6 +50,7 @@ class SpeakerVerificationPipeline(Pipeline):
|
||||
self.model_config = self.model.model_config
|
||||
self.config = self.model.other_config
|
||||
self.thr = self.config['yesOrno_thr']
|
||||
self.save_dict = {}
|
||||
|
||||
def __call__(self,
|
||||
in_audios: Union[np.ndarray, list],
|
||||
@@ -66,7 +67,9 @@ class SpeakerVerificationPipeline(Pipeline):
|
||||
embs = self.forward(wavs)
|
||||
outputs = self.postprocess(embs, in_audios, save_dir)
|
||||
if output_emb:
|
||||
return outputs, embs.numpy()
|
||||
self.save_dict['outputs'] = outputs
|
||||
self.save_dict['embs'] = embs.numpy()
|
||||
return self.save_dict
|
||||
else:
|
||||
return outputs
|
||||
|
||||
|
||||
@@ -121,6 +121,7 @@ def pipeline(task: str = None,
|
||||
ignore_file_pattern=ignore_file_pattern)
|
||||
if pipeline_name is None and kwargs.get('llm_first'):
|
||||
pipeline_name = llm_first_checker(model, model_revision)
|
||||
kwargs.pop('llm_first')
|
||||
pipeline_props = {'type': pipeline_name}
|
||||
if pipeline_name is None:
|
||||
# get default pipeline for this task
|
||||
|
||||
@@ -172,13 +172,19 @@ class CardDetectionCorrection(Pipeline):
|
||||
wh = output['wh']
|
||||
reg = output['reg']
|
||||
angle_cls = output['cls'].sigmoid_()
|
||||
ftype_cls = output['ftype'].sigmoid_()
|
||||
|
||||
bbox, inds = bbox_decode(hm, wh, reg=reg, K=self.K)
|
||||
angle_cls = decode_by_ind(
|
||||
angle_cls, inds, K=self.K).detach().cpu().numpy()
|
||||
ftype_cls = decode_by_ind(
|
||||
ftype_cls, inds,
|
||||
K=self.K).detach().cpu().numpy().astype(np.float32)
|
||||
bbox = bbox.detach().cpu().numpy()
|
||||
for i in range(bbox.shape[1]):
|
||||
bbox[0][i][9] = angle_cls[0][i]
|
||||
bbox = np.concatenate((bbox, np.expand_dims(ftype_cls, axis=-1)),
|
||||
axis=-1)
|
||||
bbox = nms(bbox, 0.3)
|
||||
bbox = bbox_post_process(bbox.copy(), [meta['c'].cpu().numpy()],
|
||||
[meta['s']], meta['out_height'],
|
||||
@@ -187,6 +193,8 @@ class CardDetectionCorrection(Pipeline):
|
||||
res = []
|
||||
angle = []
|
||||
sub_imgs = []
|
||||
ftype = []
|
||||
score = []
|
||||
for idx, box in enumerate(bbox[0]):
|
||||
if box[8] > 0.3:
|
||||
angle.append(int(box[9]))
|
||||
@@ -200,9 +208,14 @@ class CardDetectionCorrection(Pipeline):
|
||||
if angle[-1] == 3:
|
||||
sub_img = cv2.rotate(sub_img, 0)
|
||||
sub_imgs.append(sub_img)
|
||||
ftype.append(int(box[10]))
|
||||
score.append(box[8])
|
||||
|
||||
result = {
|
||||
OutputKeys.POLYGONS: np.array(res),
|
||||
OutputKeys.OUTPUT_IMGS: np.array(sub_imgs)
|
||||
OutputKeys.POLYGONS: res,
|
||||
OutputKeys.SCORES: score,
|
||||
OutputKeys.OUTPUT_IMGS: sub_imgs,
|
||||
OutputKeys.LABELS: angle,
|
||||
OutputKeys.LAYOUT: np.array(ftype)
|
||||
}
|
||||
return result
|
||||
|
||||
@@ -232,13 +232,13 @@ def nms(dets, thresh):
|
||||
keep = []
|
||||
for i in range(len(dets)):
|
||||
box = dets[i]
|
||||
if box[-1] < thresh:
|
||||
if box[8] < thresh:
|
||||
break
|
||||
max_score_index = -1
|
||||
ctx = (dets[i][0] + dets[i][2] + dets[i][4] + dets[i][6]) / 4
|
||||
cty = (dets[i][1] + dets[i][3] + dets[i][5] + dets[i][7]) / 4
|
||||
for j in range(len(dets)):
|
||||
if i == j or dets[j][-1] < thresh:
|
||||
if i == j or dets[j][8] < thresh:
|
||||
break
|
||||
x1, y1 = dets[j][0], dets[j][1]
|
||||
x2, y2 = dets[j][2], dets[j][3]
|
||||
|
||||
@@ -26,6 +26,7 @@ if TYPE_CHECKING:
|
||||
from .visual_question_answering_pipeline import VisualQuestionAnsweringPipeline
|
||||
from .video_question_answering_pipeline import VideoQuestionAnsweringPipeline
|
||||
from .videocomposer_pipeline import VideoComposerPipeline
|
||||
from .text_to_image_freeu_pipeline import FreeUTextToImagePipeline
|
||||
else:
|
||||
_import_structure = {
|
||||
'image_captioning_pipeline': ['ImageCaptioningPipeline'],
|
||||
@@ -53,7 +54,8 @@ else:
|
||||
['SOONetVideoTemporalGroundingPipeline'],
|
||||
'text_to_video_synthesis_pipeline': ['TextToVideoSynthesisPipeline'],
|
||||
'multimodal_dialogue_pipeline': ['MultimodalDialoguePipeline'],
|
||||
'videocomposer_pipeline': ['VideoComposerPipeline']
|
||||
'videocomposer_pipeline': ['VideoComposerPipeline'],
|
||||
'text_to_image_freeu_pipeline': ['FreeUTextToImagePipeline']
|
||||
}
|
||||
|
||||
import sys
|
||||
|
||||
138
modelscope/pipelines/multi_modal/text_to_image_freeu_pipeline.py
Normal file
138
modelscope/pipelines/multi_modal/text_to_image_freeu_pipeline.py
Normal file
@@ -0,0 +1,138 @@
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
import os.path
|
||||
from typing import Any, Dict, Optional, Union
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
from modelscope.metainfo import Pipelines
|
||||
from modelscope.models.multi_modal.freeu import (
|
||||
register_free_crossattn_upblock2d, register_free_upblock2d)
|
||||
from modelscope.outputs import OutputKeys
|
||||
from modelscope.pipelines import pipeline
|
||||
from modelscope.pipelines.base import Pipeline
|
||||
from modelscope.pipelines.builder import PIPELINES
|
||||
from modelscope.utils.constant import Tasks
|
||||
from modelscope.utils.logger import get_logger
|
||||
|
||||
logger = get_logger()
|
||||
|
||||
__all__ = ['FreeUTextToImagePipeline']
|
||||
|
||||
|
||||
@PIPELINES.register_module(
|
||||
Tasks.text_to_image_synthesis,
|
||||
module_name=Pipelines.freeu_stable_diffusion_text2image)
|
||||
class FreeUTextToImagePipeline(Pipeline):
|
||||
|
||||
def __init__(self, model=str, preprocessor=None, **kwargs):
|
||||
""" FreeU Text to Image Pipeline.
|
||||
|
||||
Examples:
|
||||
|
||||
>>> import cv2
|
||||
>>> from modelscope.pipelines import pipeline
|
||||
>>> from modelscope.utils.constant import Tasks
|
||||
|
||||
>>> prompt = "a photo of a running corgi" # prompt
|
||||
>>> output_image_path = './result.png'
|
||||
>>> inputs = {'prompt': prompt}
|
||||
>>>
|
||||
>>> pipe = pipeline(
|
||||
>>> Tasks.text_to_image_synthesis,
|
||||
>>> model='damo/multi-modal_freeu_stable_diffusion',
|
||||
>>> base_model='AI-ModelScope/stable-diffusion-v1-5',
|
||||
>>> )
|
||||
>>>
|
||||
>>> output = pipe(inputs)['output_imgs']
|
||||
>>> cv2.imwrite(output_image_path, output)
|
||||
>>> print('pipeline: the output image path is {}'.format(output_image_path))
|
||||
"""
|
||||
super().__init__(model=model, preprocessor=preprocessor, **kwargs)
|
||||
|
||||
torch_dtype = kwargs.get('torch_dtype', torch.float32)
|
||||
self._device = getattr(
|
||||
kwargs, 'device',
|
||||
torch.device('cuda' if torch.cuda.is_available() else 'cpu'))
|
||||
base_model = kwargs.get(
|
||||
'base_model', 'AI-ModelScope/stable-diffusion-v1-5') # default 1.5
|
||||
self.freeu_params = kwargs.get('freeu_params', {
|
||||
'b1': 1.5,
|
||||
'b2': 1.6,
|
||||
's1': 0.9,
|
||||
's2': 0.2
|
||||
}) # default
|
||||
|
||||
logger.info('load freeu stable diffusion text to image pipeline done')
|
||||
self.pipeline = pipeline(
|
||||
task=Tasks.text_to_image_synthesis,
|
||||
model=base_model,
|
||||
torch_dtype=torch_dtype,
|
||||
device=self._device).pipeline
|
||||
|
||||
def preprocess(self, inputs: Dict[str, Any], **kwargs) -> Dict[str, Any]:
|
||||
return inputs
|
||||
|
||||
def forward(self, inputs: Dict[str, Any], **kwargs) -> Dict[str, Any]:
|
||||
"""
|
||||
Inputs Args:
|
||||
prompt (`str` or `List[str]`, *optional*):
|
||||
The prompt or prompts to guide the image generation. If not defined, one has to pass `prompt_embeds`.
|
||||
instead.
|
||||
height (`int`, *optional*, defaults to self.unet.config.sample_size * self.vae_scale_factor):
|
||||
The height in pixels of the generated image.
|
||||
width (`int`, *optional*, defaults to self.unet.config.sample_size * self.vae_scale_factor):
|
||||
The width in pixels of the generated image.
|
||||
num_inference_steps (`int`, *optional*, defaults to 50):
|
||||
The number of denoising steps. More denoising steps usually lead to a higher quality image at the
|
||||
expense of slower inference.
|
||||
guidance_scale (`float`, *optional*, defaults to 7.5):
|
||||
Guidance scale as defined in [Classifier-Free Diffusion Guidance](https://arxiv.org/abs/2207.12598).
|
||||
`guidance_scale` is defined as `w` of equation 2. of [Imagen
|
||||
Paper](https://arxiv.org/pdf/2205.11487.pdf). Guidance scale is enabled by setting `guidance_scale >
|
||||
1`. Higher guidance scale encourages to generate images that are closely linked to the text `prompt`,
|
||||
usually at the expense of lower image quality.
|
||||
negative_prompt (`str` or `List[str]`, *optional*):
|
||||
The prompt or prompts not to guide the image generation. If not defined, one has to pass
|
||||
`negative_prompt_embeds` instead. Ignored when not using guidance (i.e., ignored if `guidance_scale` is
|
||||
less than `1`).
|
||||
num_images_per_prompt (`int`, *optional*, defaults to 1):
|
||||
The number of images to generate per prompt.
|
||||
eta (`float`, *optional*, defaults to 0.0):
|
||||
Corresponds to parameter eta (η) in the DDIM paper: https://arxiv.org/abs/2010.02502. Only applies to
|
||||
[`schedulers.DDIMScheduler`], will be ignored for others.
|
||||
generator (`torch.Generator` or `List[torch.Generator]`, *optional*):
|
||||
One or a list of [torch generator(s)](https://pytorch.org/docs/stable/generated/torch.Generator.html)
|
||||
to make generation deterministic.
|
||||
latents (`torch.FloatTensor`, *optional*):
|
||||
Pre-generated noisy latents, sampled from a Gaussian distribution, to be used as inputs for image
|
||||
generation. Can be used to tweak the same generation with different prompts. If not provided, a latents
|
||||
tensor will ge generated by sampling using the supplied random `generator`.
|
||||
"""
|
||||
if not isinstance(inputs, dict):
|
||||
raise ValueError(
|
||||
f'Expected the input to be a dictionary, but got {type(inputs)}'
|
||||
)
|
||||
# -------- freeu block registration
|
||||
register_free_upblock2d(self.pipeline, **self.freeu_params)
|
||||
register_free_crossattn_upblock2d(self.pipeline, **self.freeu_params)
|
||||
# -------- freeu block registration
|
||||
|
||||
output = self.pipeline(
|
||||
prompt=inputs.get('prompt'),
|
||||
height=inputs.get('height'),
|
||||
width=inputs.get('width'),
|
||||
num_inference_steps=inputs.get('num_inference_steps', 50),
|
||||
guidance_scale=inputs.get('guidance_scale', 7.5),
|
||||
negative_prompt=inputs.get('negative_prompt'),
|
||||
num_images_per_prompt=inputs.get('num_images_per_prompt', 1),
|
||||
eta=inputs.get('eta', 0.0),
|
||||
generator=inputs.get('generator'),
|
||||
latents=inputs.get('latents'),
|
||||
).images[0]
|
||||
|
||||
return {'output_tensor': output}
|
||||
|
||||
def postprocess(self, inputs: Dict[str, Any], **kwargs) -> Dict[str, Any]:
|
||||
output_img = np.array(inputs['output_tensor'])
|
||||
return {OutputKeys.OUTPUT_IMGS: output_img[:, :, ::-1]}
|
||||
@@ -1,14 +1,19 @@
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
|
||||
from typing import Any, Dict
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
import torch
|
||||
|
||||
from modelscope.metainfo import Preprocessors
|
||||
from modelscope.preprocessors import Preprocessor
|
||||
from modelscope.preprocessors.builder import PREPROCESSORS
|
||||
from modelscope.utils.constant import Fields, ModeKeys
|
||||
from modelscope.utils.hub import get_model_type
|
||||
from modelscope.utils.logger import get_logger
|
||||
from .transformers_tokenizer import NLPTokenizer
|
||||
|
||||
logger = get_logger()
|
||||
|
||||
|
||||
@PREPROCESSORS.register_module(
|
||||
Fields.nlp, module_name=Preprocessors.sentence_embedding)
|
||||
@@ -46,9 +51,32 @@ class SentenceEmbeddingTransformersPreprocessor(Preprocessor):
|
||||
self.max_length = max_length
|
||||
if model_dir is not None:
|
||||
model_type = get_model_type(model_dir)
|
||||
# we could add `boq/bod` token/prompt and `eoq/eod` token if they exist when tokenizing.
|
||||
for k in ('boq', 'eoq', 'bod', 'eod'):
|
||||
setattr(self, k, kwargs.pop(k, None))
|
||||
self.nlp_tokenizer = NLPTokenizer(
|
||||
model_dir, model_type, use_fast=use_fast, tokenize_kwargs=kwargs)
|
||||
super().__init__(mode=mode)
|
||||
tokenizer = self.nlp_tokenizer.tokenizer
|
||||
# For tokenizers like bloom
|
||||
if tokenizer.padding_side != 'right':
|
||||
# weighted mean pooling need pad right
|
||||
logger.warning(
|
||||
f'Change tokenizer.padding_side from {tokenizer.padding_side} to right'
|
||||
)
|
||||
tokenizer.padding_side = 'right'
|
||||
# For decoder-only tokenizers
|
||||
if tokenizer.pad_token is None:
|
||||
logger.warning(
|
||||
f'Set tokenizer.pad_token as eos_token {tokenizer.eos_token}')
|
||||
tokenizer.pad_token = tokenizer.eos_token
|
||||
# Currently eos is single token, we can extend to prompt later.
|
||||
for k in ('eoq', 'eod'):
|
||||
v = getattr(self, k, None)
|
||||
if v is not None:
|
||||
v = tokenizer.convert_tokens_to_ids(v)
|
||||
setattr(self, k + '_id', v)
|
||||
self.pad_id = tokenizer.convert_tokens_to_ids(tokenizer.pad_token)
|
||||
|
||||
def __call__(self,
|
||||
data: Dict,
|
||||
@@ -81,13 +109,80 @@ class SentenceEmbeddingTransformersPreprocessor(Preprocessor):
|
||||
if 'return_tensors' not in kwargs:
|
||||
kwargs[
|
||||
'return_tensors'] = 'pt' if self.mode == ModeKeys.INFERENCE else None
|
||||
query_inputs = self.nlp_tokenizer(
|
||||
source_sentences, padding=padding, truncation=truncation, **kwargs)
|
||||
query_inputs = self.tokenize(
|
||||
source_sentences,
|
||||
is_query=True,
|
||||
padding=padding,
|
||||
truncation=truncation,
|
||||
**kwargs)
|
||||
tokenized_inputs = {'query': query_inputs, 'docs': None}
|
||||
if compare_sentences is not None and len(compare_sentences) > 0:
|
||||
tokenized_inputs['docs'] = self.nlp_tokenizer(
|
||||
tokenized_inputs['docs'] = self.tokenize(
|
||||
compare_sentences,
|
||||
is_query=kwargs.get('symmetric', False),
|
||||
padding=padding,
|
||||
truncation=truncation,
|
||||
**kwargs)
|
||||
return tokenized_inputs
|
||||
|
||||
def tokenize(self, texts, is_query=True, return_tensors=None, **kwargs):
|
||||
"""Tokenize raw texts, add `boq/bod` token/prompt and `eoq/eod` token if they exist.
|
||||
|
||||
Args:
|
||||
`texts` List[str]: texts to tokenize,
|
||||
Example:
|
||||
["how long it take to get a master's degree"]
|
||||
`is_query` bool: whether the input text(s) is query.
|
||||
`return_tensors` str: the `return_tensors` argument to tokenizer.
|
||||
Returns:
|
||||
Dict[str, Any]: the preprocessed data
|
||||
"""
|
||||
if is_query:
|
||||
bos, eos_id = self.boq, self.eoq_id
|
||||
else:
|
||||
bos, eos_id = self.bod, self.eod_id
|
||||
if bos is not None:
|
||||
# bos can be prompt
|
||||
texts = [bos + t for t in texts]
|
||||
encoding = self.nlp_tokenizer(
|
||||
texts, return_tensors=return_tensors, **kwargs)
|
||||
if eos_id is not None:
|
||||
if return_tensors == 'pt':
|
||||
self.add_eos_pt(encoding, eos_id)
|
||||
else:
|
||||
self.add_eos(encoding, eos_id)
|
||||
return encoding
|
||||
|
||||
def add_eos_pt(self, encoding: Dict[str, torch.Tensor], eos: int):
|
||||
"""Add `eos` token id to the end of each sequence."""
|
||||
input_ids, attn_mask = encoding['input_ids'], encoding[
|
||||
'attention_mask']
|
||||
batch = torch.arange(input_ids.size(0))
|
||||
length = attn_mask.sum(-1)
|
||||
|
||||
if input_ids.size(1) < self.max_length:
|
||||
ones = input_ids.new_ones(input_ids.size(0), 1)
|
||||
attn_mask = torch.cat((ones, attn_mask), dim=1)
|
||||
padding = ones * self.pad_id
|
||||
input_ids = torch.cat((input_ids, padding), dim=1)
|
||||
eos_index = length
|
||||
else:
|
||||
eos_index = torch.clamp(length, max=self.max_length - 1)
|
||||
attn_mask[batch, eos_index] = 1
|
||||
input_ids[batch, eos_index] = eos
|
||||
encoding['input_ids'], encoding[
|
||||
'attention_mask'] = input_ids, attn_mask
|
||||
return
|
||||
|
||||
def add_eos(self, encoding: Dict[str, list], eos: int):
|
||||
"""Add `eos` token id to the end of each sequence."""
|
||||
for ids, mask in zip(encoding['input_ids'],
|
||||
encoding['attention_mask']):
|
||||
if len(mask) < self.max_length:
|
||||
ids.append(eos)
|
||||
mask.append(1)
|
||||
else:
|
||||
last = min(sum(mask), self.max_length - 1)
|
||||
ids[last] = eos
|
||||
mask[last] = 1
|
||||
return
|
||||
|
||||
@@ -145,6 +145,19 @@
|
||||
"image":"http://modelscope.oss-cn-beijing.aliyuncs.com/demo/images/image_salient_detection.jpg"
|
||||
}
|
||||
},
|
||||
"sentence-embedding":{
|
||||
"input": {
|
||||
"source_sentence":[
|
||||
"吃完海鲜可以喝牛奶吗?"
|
||||
],
|
||||
"sentences_to_compare":[
|
||||
"不可以,早晨喝牛奶不科学",
|
||||
"吃了海鲜后是不能再喝牛奶的,因为牛奶中含得有维生素C,如果海鲜喝牛奶一起服用会对人体造成一定的伤害",
|
||||
"吃海鲜是不能同时喝牛奶吃水果,这个至少间隔6小时以上才可以。",
|
||||
"吃海鲜是不可以吃柠檬的因为其中的维生素C会和海鲜中的矿物质形成砷"
|
||||
]
|
||||
}
|
||||
},
|
||||
"shop-segmentation":{
|
||||
"input":{
|
||||
"image":"http://modelscope.oss-cn-beijing.aliyuncs.com/demo/images/shop_segmentation.jpg"
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
# Make sure to modify __release_datetime__ to release time when making official release.
|
||||
__version__ = '1.9.3'
|
||||
__version__ = '1.9.4'
|
||||
# default release datetime for branches under active development is set
|
||||
# to be a time far-far-away-into-the-future
|
||||
__release_datetime__ = '2099-09-06 00:00:00'
|
||||
|
||||
@@ -25,7 +25,8 @@ class ControllableImageGenerationTest(unittest.TestCase):
|
||||
'prompt': 'flower'
|
||||
}
|
||||
|
||||
@unittest.skipUnless(test_level() >= 0, 'skip test in current test level')
|
||||
@unittest.skipUnless(test_level() >= 2,
|
||||
'skip test for huggingface model download issue.')
|
||||
def test_run_with_model_from_modelhub(self):
|
||||
output_image_path = tempfile.NamedTemporaryFile(suffix='.png').name
|
||||
control_types = [
|
||||
|
||||
@@ -21,6 +21,7 @@ class SentenceEmbeddingTest(unittest.TestCase):
|
||||
medical_tiny_model_id = 'damo/nlp_corom_sentence-embedding_chinese-tiny-medical'
|
||||
general_base_model_id = 'damo/nlp_corom_sentence-embedding_chinese-base'
|
||||
general_tiny_model_id = 'damo/nlp_corom_sentence-embedding_chinese-tiny'
|
||||
bloom_model_id = 'damo/udever-bloom-7b1'
|
||||
|
||||
inputs = {
|
||||
'source_sentence': ["how long it take to get a master's degree"],
|
||||
@@ -154,6 +155,14 @@ class SentenceEmbeddingTest(unittest.TestCase):
|
||||
print()
|
||||
print(f'pipeline2: {pipeline2(input=self.medical_inputs1)}')
|
||||
|
||||
@unittest.skipUnless(test_level() >= 0, 'skip test in current test level')
|
||||
def test_run_with_bloom_model_from_modelhub(self):
|
||||
model = Model.from_pretrained(self.bloom_model_id)
|
||||
tokenizer = SentenceEmbeddingTransformersPreprocessor(model.model_dir)
|
||||
pipeline_ins = pipeline(
|
||||
task=Tasks.sentence_embedding, model=model, preprocessor=tokenizer)
|
||||
print(pipeline_ins(input=self.inputs))
|
||||
|
||||
@unittest.skipUnless(test_level() >= 0, 'skip test in current test level')
|
||||
def test_run_with_model_from_modelhub(self):
|
||||
model = Model.from_pretrained(self.model_id)
|
||||
|
||||
@@ -23,8 +23,10 @@ class SpeakerVerificationTest(unittest.TestCase):
|
||||
campplus_voxceleb_16k_model_id = 'damo/speech_campplus_sv_en_voxceleb_16k'
|
||||
rdino_voxceleb_16k_model_id = 'damo/speech_rdino_ecapa_tdnn_sv_en_voxceleb_16k'
|
||||
speaker_change_locating_cn_model_id = 'damo/speech_campplus-transformer_scl_zh-cn_16k-common'
|
||||
speaker_change_lcoating_xvector_cn_model_id = 'damo/speech_xvector_transformer_scl_zh-cn_16k-common'
|
||||
eres2net_voxceleb_16k_model_id = 'damo/speech_eres2net_sv_en_voxceleb_16k'
|
||||
speaker_diarization_model_id = 'damo/speech_campplus_speaker-diarization_common'
|
||||
speaker_diarization_eres2net_model_id = 'damo/speech_eres2net-large_speaker-diarization_common'
|
||||
lre_campplus_en_cn_16k_model_id = 'damo/speech_campplus_lre_en-cn_16k'
|
||||
lre_eres2net_base_en_cn_16k_model_id = 'damo/speech_eres2net_base_lre_en-cn_16k'
|
||||
lre_eres2net_large_en_cn_16k_model_id = 'damo/speech_eres2net_large_lre_en-cn_16k'
|
||||
@@ -123,6 +125,17 @@ class SpeakerVerificationTest(unittest.TestCase):
|
||||
print(result)
|
||||
self.assertTrue(OutputKeys.TEXT in result)
|
||||
|
||||
@unittest.skipUnless(test_level() >= 0, 'skip test in current test level')
|
||||
def test_run_with_speaker_change_locating_xvector_cn_16k(self):
|
||||
logger.info(
|
||||
'Run speaker change locating for xvector-transformer model')
|
||||
result = self.run_pipeline(
|
||||
model_id=self.speaker_change_lcoating_xvector_cn_model_id,
|
||||
task=Tasks.speaker_diarization,
|
||||
audios=SCL_EXAMPLE_WAV)
|
||||
print(result)
|
||||
self.assertTrue(OutputKeys.TEXT in result)
|
||||
|
||||
@unittest.skipUnless(test_level() >= 0, 'skip test in current test level')
|
||||
def test_run_with_speaker_verification_eres2net_voxceleb_16k(self):
|
||||
logger.info('Run speaker verification for eres2net_voxceleb_16k model')
|
||||
@@ -140,7 +153,7 @@ class SpeakerVerificationTest(unittest.TestCase):
|
||||
result = self.run_pipeline(
|
||||
model_id=self.eres2net_aug_zh_cn_16k_common_model_id,
|
||||
audios=[SPEAKER1_A_EN_16K_WAV, SPEAKER1_B_EN_16K_WAV],
|
||||
model_revision='v1.0.4')
|
||||
model_revision='v1.0.5')
|
||||
print(result)
|
||||
self.assertTrue(OutputKeys.SCORE in result)
|
||||
|
||||
@@ -154,6 +167,16 @@ class SpeakerVerificationTest(unittest.TestCase):
|
||||
print(result)
|
||||
self.assertTrue(OutputKeys.TEXT in result)
|
||||
|
||||
@unittest.skipUnless(test_level() >= 0, 'skip test in current test level')
|
||||
def test_run_with_eres2net_speaker_diarization_common(self):
|
||||
logger.info('Run eres2net speaker diarization task')
|
||||
result = self.run_pipeline(
|
||||
model_id=self.speaker_diarization_eres2net_model_id,
|
||||
task=Tasks.speaker_diarization,
|
||||
audios=SD_EXAMPLE_WAV)
|
||||
print(result)
|
||||
self.assertTrue(OutputKeys.TEXT in result)
|
||||
|
||||
@unittest.skipUnless(test_level() >= 0, 'skip test in current test level')
|
||||
def test_run_with_language_recognition_campplus_en_cn_16k(self):
|
||||
logger.info('Run language recognition for campplus_en_cn_16k')
|
||||
|
||||
57
tests/pipelines/test_text_to_image_freeu.py
Normal file
57
tests/pipelines/test_text_to_image_freeu.py
Normal file
@@ -0,0 +1,57 @@
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
|
||||
import unittest
|
||||
|
||||
import cv2
|
||||
|
||||
from modelscope.hub.snapshot_download import snapshot_download
|
||||
from modelscope.outputs import OutputKeys
|
||||
from modelscope.pipelines import pipeline
|
||||
from modelscope.pipelines.multi_modal import FreeUTextToImagePipeline
|
||||
from modelscope.utils.constant import Tasks
|
||||
from modelscope.utils.test_utils import test_level
|
||||
|
||||
|
||||
class ImageEditingTest(unittest.TestCase):
|
||||
|
||||
def setUp(self) -> None:
|
||||
self.task = Tasks.text_to_image_synthesis
|
||||
self.model_id = 'damo/multi-modal_freeu_stable_diffusion'
|
||||
prompt = 'a photo of a running corgi' # prompt
|
||||
self.inputs = {'prompt': prompt}
|
||||
self.output_image_path = './result.png'
|
||||
self.base_model = 'AI-ModelScope/stable-diffusion-v2-1'
|
||||
self.freeu_params = {
|
||||
'b1': 1.4,
|
||||
'b2': 1.6,
|
||||
's1': 0.9,
|
||||
's2': 0.2
|
||||
} # for SD2.1
|
||||
|
||||
@unittest.skipUnless(test_level() >= 2, 'skip test in current test level')
|
||||
def test_run_by_direct_model_download(self):
|
||||
cache_path = snapshot_download(self.model_id)
|
||||
pipeline = FreeUTextToImagePipeline(cache_path)
|
||||
pipeline.group_key = self.task
|
||||
synthesized_img = pipeline(
|
||||
input=self.inputs)[OutputKeys.OUTPUT_IMGS] # BGR
|
||||
cv2.imwrite(self.output_image_path, synthesized_img)
|
||||
print('FreeU pipeline: the synthesized image path is {}'.format(
|
||||
self.output_image_path))
|
||||
|
||||
@unittest.skipUnless(test_level() >= 1, 'skip test in current test level')
|
||||
def test_run_with_model_name(self):
|
||||
pipeline_ins = pipeline(
|
||||
task=Tasks.text_to_image_synthesis,
|
||||
model=self.model_id,
|
||||
base_model=self.base_model,
|
||||
freeu_params=self.freeu_params)
|
||||
synthesized_img = pipeline_ins(
|
||||
self.inputs)[OutputKeys.OUTPUT_IMGS] # BGR
|
||||
cv2.imwrite(self.output_image_path, synthesized_img)
|
||||
print('FreeU pipeline: the synthesized image path is {}'.format(
|
||||
self.output_image_path))
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
@@ -18,10 +18,10 @@ class HFUtilTest(unittest.TestCase):
|
||||
|
||||
def test_auto_tokenizer(self):
|
||||
tokenizer = AutoTokenizer.from_pretrained(
|
||||
'baichuan-inc/Baichuan-13B-Chat',
|
||||
'baichuan-inc/Baichuan2-7B-Chat',
|
||||
trust_remote_code=True,
|
||||
revision='v1.0.3')
|
||||
self.assertEqual(tokenizer.vocab_size, 64000)
|
||||
self.assertEqual(tokenizer.vocab_size, 125696)
|
||||
self.assertEqual(tokenizer.model_max_length, 4096)
|
||||
self.assertFalse(tokenizer.is_fast)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user