mirror of
https://github.com/modelscope/modelscope.git
synced 2026-09-01 19:49:03 +02:00
add plug mental model
Link: https://code.alibaba-inc.com/Ali-MaaS/MaaS-lib/codereview/11549696 * add plug mental model code * add test pipeline and fix annotation format bugs
This commit is contained in:
@@ -123,6 +123,7 @@ class Models(object):
|
||||
unite = 'unite'
|
||||
megatron_bert = 'megatron-bert'
|
||||
use = 'user-satisfaction-estimation'
|
||||
plug_mental = 'plug-mental'
|
||||
|
||||
# audio models
|
||||
sambert_hifigan = 'sambert-hifigan'
|
||||
|
||||
@@ -61,6 +61,8 @@ if TYPE_CHECKING:
|
||||
from .bloom import BloomModel
|
||||
from .unite import UniTEModel
|
||||
from .use import UserSatisfactionEstimation
|
||||
from .plug_mental import (PlugMentalConfig, PlugMentalModel,
|
||||
PlugMentalForSequenceClassification)
|
||||
else:
|
||||
_import_structure = {
|
||||
'backbones': ['SbertModel'],
|
||||
@@ -127,7 +129,12 @@ else:
|
||||
'gpt_neo': ['GPTNeoModel'],
|
||||
'bloom': ['BloomModel'],
|
||||
'unite': ['UniTEModel'],
|
||||
'use': ['UserSatisfactionEstimation']
|
||||
'use': ['UserSatisfactionEstimation'],
|
||||
'plug_mental': [
|
||||
'PlugMentalConfig',
|
||||
'PlugMentalModel',
|
||||
'PlugMentalForSequenceClassification',
|
||||
]
|
||||
}
|
||||
|
||||
import sys
|
||||
|
||||
39
modelscope/models/nlp/plug_mental/__init__.py
Normal file
39
modelscope/models/nlp/plug_mental/__init__.py
Normal file
@@ -0,0 +1,39 @@
|
||||
# Copyright 2021-2022 The Alibaba DAMO NLP Team Authors.
|
||||
# All rights reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from modelscope.utils.import_utils import LazyImportModule
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from .backbone import (PlugMentalModel, PlugMentalPreTrainedModel)
|
||||
from .configuration import PlugMentalConfig
|
||||
from .text_classification import PlugMentalForSequenceClassification
|
||||
else:
|
||||
_import_structure = {
|
||||
'backbone': ['PlugMentalModel', 'PlugMentalPreTrainedModel'],
|
||||
'configuration': ['PlugMentalConfig'],
|
||||
'text_classification': ['PlugMentalForSequenceClassification'],
|
||||
}
|
||||
|
||||
import sys
|
||||
|
||||
sys.modules[__name__] = LazyImportModule(
|
||||
__name__,
|
||||
globals()['__file__'],
|
||||
_import_structure,
|
||||
module_spec=__spec__,
|
||||
extra_objects={},
|
||||
)
|
||||
168
modelscope/models/nlp/plug_mental/adv_utils.py
Normal file
168
modelscope/models/nlp/plug_mental/adv_utils.py
Normal file
@@ -0,0 +1,168 @@
|
||||
# Copyright 2021-2022 The Alibaba DAMO NLP Team Authors.
|
||||
# All rights reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import torch
|
||||
from torch import nn
|
||||
|
||||
from modelscope.utils.logger import get_logger
|
||||
|
||||
logger = get_logger()
|
||||
|
||||
|
||||
def _symmetric_kl_div(logits1, logits2, attention_mask=None):
|
||||
"""
|
||||
Calclate two logits' the KL div value symmetrically.
|
||||
:param logits1: The first logit.
|
||||
:param logits2: The second logit.
|
||||
:param attention_mask: An optional attention_mask which is used to mask some element out.
|
||||
This is usually useful in token_classification tasks.
|
||||
If the shape of logits is [N1, N2, ... Nn, D], the shape of attention_mask should be [N1, N2, ... Nn]
|
||||
:return: The mean loss.
|
||||
"""
|
||||
labels_num = logits1.shape[-1]
|
||||
KLDiv = nn.KLDivLoss(reduction='none')
|
||||
loss = torch.sum(
|
||||
KLDiv(nn.LogSoftmax(dim=-1)(logits1),
|
||||
nn.Softmax(dim=-1)(logits2)),
|
||||
dim=-1) + torch.sum(
|
||||
KLDiv(nn.LogSoftmax(dim=-1)(logits2),
|
||||
nn.Softmax(dim=-1)(logits1)),
|
||||
dim=-1)
|
||||
if attention_mask is not None:
|
||||
loss = torch.sum(
|
||||
loss * attention_mask) / torch.sum(attention_mask) / labels_num
|
||||
else:
|
||||
loss = torch.mean(loss) / labels_num
|
||||
return loss
|
||||
|
||||
|
||||
def compute_adv_loss(embedding,
|
||||
model,
|
||||
ori_logits,
|
||||
ori_loss,
|
||||
adv_grad_factor,
|
||||
adv_bound=None,
|
||||
sigma=5e-6,
|
||||
**kwargs):
|
||||
"""
|
||||
Calculate the adv loss of the model.
|
||||
:param embedding: Original sentense embedding
|
||||
:param model: The model, or the forward function(including decoder/classifier),
|
||||
accept kwargs as input, output logits
|
||||
:param ori_logits: The original logits outputed from the model function
|
||||
:param ori_loss: The original loss
|
||||
:param adv_grad_factor: This factor will be multipled by the KL loss grad and then the result will be added to
|
||||
the original embedding.
|
||||
More details please check:https://arxiv.org/abs/1908.04577
|
||||
The range of this value always be 1e-3~1e-7
|
||||
:param adv_bound: adv_bound is used to cut the top and the bottom bound of the produced embedding.
|
||||
If not proveded, 2 * sigma will be used as the adv_bound factor
|
||||
:param sigma: The std factor used to produce a 0 mean normal distribution.
|
||||
If adv_bound not proveded, 2 * sigma will be used as the adv_bound factor
|
||||
:param kwargs: the input param used in model function
|
||||
:return: The original loss adds the adv loss
|
||||
"""
|
||||
adv_bound = adv_bound if adv_bound is not None else 2 * sigma
|
||||
embedding_1 = embedding + embedding.data.new(embedding.size()).normal_(
|
||||
0, sigma) # 95% in +- 1e-5
|
||||
kwargs.pop('input_ids')
|
||||
if 'inputs_embeds' in kwargs:
|
||||
kwargs.pop('inputs_embeds')
|
||||
with_attention_mask = False if 'with_attention_mask' not in kwargs else kwargs[
|
||||
'with_attention_mask']
|
||||
attention_mask = kwargs['attention_mask']
|
||||
if not with_attention_mask:
|
||||
attention_mask = None
|
||||
if 'with_attention_mask' in kwargs:
|
||||
kwargs.pop('with_attention_mask')
|
||||
outputs = model(**kwargs, inputs_embeds=embedding_1)
|
||||
v1_logits = outputs.logits
|
||||
loss = _symmetric_kl_div(ori_logits, v1_logits, attention_mask)
|
||||
emb_grad = torch.autograd.grad(loss, embedding_1)[0].data
|
||||
emb_grad_norm = emb_grad.norm(
|
||||
dim=2, keepdim=True, p=float('inf')).max(
|
||||
1, keepdim=True)[0]
|
||||
is_nan = torch.any(torch.isnan(emb_grad_norm))
|
||||
if is_nan:
|
||||
logger.warning('Nan occured when calculating adv loss.')
|
||||
return ori_loss
|
||||
emb_grad = emb_grad / (emb_grad_norm + 1e-6)
|
||||
embedding_2 = embedding_1 + adv_grad_factor * emb_grad
|
||||
embedding_2 = torch.max(embedding_1 - adv_bound, embedding_2)
|
||||
embedding_2 = torch.min(embedding_1 + adv_bound, embedding_2)
|
||||
outputs = model(**kwargs, inputs_embeds=embedding_2)
|
||||
adv_logits = outputs.logits
|
||||
adv_loss = _symmetric_kl_div(ori_logits, adv_logits, attention_mask)
|
||||
return ori_loss + adv_loss
|
||||
|
||||
|
||||
def compute_adv_loss_pair(embedding,
|
||||
model,
|
||||
start_logits,
|
||||
end_logits,
|
||||
ori_loss,
|
||||
adv_grad_factor,
|
||||
adv_bound=None,
|
||||
sigma=5e-6,
|
||||
**kwargs):
|
||||
"""
|
||||
Calculate the adv loss of the model. This function is used in the pair logits scenerio.
|
||||
:param embedding: Original sentense embedding
|
||||
:param model: The model, or the forward function(including decoder/classifier),
|
||||
accept kwargs as input, output logits
|
||||
:param start_logits: The original start logits outputed from the model function
|
||||
:param end_logits: The original end logits outputed from the model function
|
||||
:param ori_loss: The original loss
|
||||
:param adv_grad_factor: This factor will be multipled by the KL loss grad and then the result will be added to
|
||||
the original embedding.
|
||||
More details please check:https://arxiv.org/abs/1908.04577
|
||||
The range of this value always be 1e-3~1e-7
|
||||
:param adv_bound: adv_bound is used to cut the top and the bottom bound of the produced embedding.
|
||||
If not proveded, 2 * sigma will be used as the adv_bound factor
|
||||
:param sigma: The std factor used to produce a 0 mean normal distribution.
|
||||
If adv_bound not proveded, 2 * sigma will be used as the adv_bound factor
|
||||
:param kwargs: the input param used in model function
|
||||
:return: The original loss adds the adv loss
|
||||
"""
|
||||
adv_bound = adv_bound if adv_bound is not None else 2 * sigma
|
||||
embedding_1 = embedding + embedding.data.new(embedding.size()).normal_(
|
||||
0, sigma) # 95% in +- 1e-5
|
||||
kwargs.pop('input_ids')
|
||||
if 'inputs_embeds' in kwargs:
|
||||
kwargs.pop('inputs_embeds')
|
||||
outputs = model(**kwargs, inputs_embeds=embedding_1)
|
||||
v1_logits_start, v1_logits_end = outputs.logits
|
||||
loss = _symmetric_kl_div(start_logits,
|
||||
v1_logits_start) + _symmetric_kl_div(
|
||||
end_logits, v1_logits_end)
|
||||
loss = loss / 2
|
||||
emb_grad = torch.autograd.grad(loss, embedding_1)[0].data
|
||||
emb_grad_norm = emb_grad.norm(
|
||||
dim=2, keepdim=True, p=float('inf')).max(
|
||||
1, keepdim=True)[0]
|
||||
is_nan = torch.any(torch.isnan(emb_grad_norm))
|
||||
if is_nan:
|
||||
logger.warning('Nan occured when calculating pair adv loss.')
|
||||
return ori_loss
|
||||
emb_grad = emb_grad / emb_grad_norm
|
||||
embedding_2 = embedding_1 + adv_grad_factor * emb_grad
|
||||
embedding_2 = torch.max(embedding_1 - adv_bound, embedding_2)
|
||||
embedding_2 = torch.min(embedding_1 + adv_bound, embedding_2)
|
||||
outputs = model(**kwargs, inputs_embeds=embedding_2)
|
||||
adv_logits_start, adv_logits_end = outputs.logits
|
||||
adv_loss = _symmetric_kl_div(start_logits,
|
||||
adv_logits_start) + _symmetric_kl_div(
|
||||
end_logits, adv_logits_end)
|
||||
return ori_loss + adv_loss
|
||||
1077
modelscope/models/nlp/plug_mental/backbone.py
Executable file
1077
modelscope/models/nlp/plug_mental/backbone.py
Executable file
File diff suppressed because it is too large
Load Diff
142
modelscope/models/nlp/plug_mental/configuration.py
Normal file
142
modelscope/models/nlp/plug_mental/configuration.py
Normal file
@@ -0,0 +1,142 @@
|
||||
# Copyright 2021-2022 The Alibaba DAMO NLP Team Authors.
|
||||
# Copyright 2018 The Google AI Language Team Authors and The HuggingFace Inc. team.
|
||||
# Copyright (c) 2018, NVIDIA CORPORATION. All rights reserved.
|
||||
# All rights reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
""" PLUG mental model configuration, mainly copied from :class:`~transformers.BertConfig` """
|
||||
from transformers import PretrainedConfig
|
||||
|
||||
from modelscope.utils import logger as logging
|
||||
|
||||
logger = logging.get_logger()
|
||||
|
||||
|
||||
class PlugMentalConfig(PretrainedConfig):
|
||||
r"""
|
||||
This is the configuration class to store the configuration
|
||||
of a :class:`~modelscope.models.nlp.plug_mental.PLugMentalModel`.
|
||||
It is used to instantiate a PlugMental model according to the specified arguments.
|
||||
|
||||
Configuration objects inherit from :class:`~transformers.PretrainedConfig` and can be used to control the model
|
||||
outputs. Read the documentation from :class:`~transformers.PretrainedConfig` for more information.
|
||||
|
||||
|
||||
Args:
|
||||
vocab_size (:obj:`int`, `optional`, defaults to 30522):
|
||||
Vocabulary size of the BERT model. Defines the number of different tokens that can be represented by the
|
||||
:obj:`inputs_ids` passed when calling :class:`~transformers.BertModel` or
|
||||
:class:`~transformers.TFBertModel`.
|
||||
hidden_size (:obj:`int`, `optional`, defaults to 768):
|
||||
Dimensionality of the encoder layers and the pooler layer.
|
||||
num_hidden_layers (:obj:`int`, `optional`, defaults to 12):
|
||||
Number of hidden layers in the Transformer encoder.
|
||||
num_attention_heads (:obj:`int`, `optional`, defaults to 12):
|
||||
Number of attention heads for each attention layer in the Transformer encoder.
|
||||
intermediate_size (:obj:`int`, `optional`, defaults to 3072):
|
||||
Dimensionality of the "intermediate" (often named feed-forward) layer in the Transformer encoder.
|
||||
hidden_act (:obj:`str` or :obj:`Callable`, `optional`, defaults to :obj:`"gelu"`):
|
||||
The non-linear activation function (function or string) in the encoder and pooler. If string,
|
||||
:obj:`"gelu"`, :obj:`"relu"`, :obj:`"silu"` and :obj:`"gelu_new"` are supported.
|
||||
hidden_dropout_prob (:obj:`float`, `optional`, defaults to 0.1):
|
||||
The dropout probability for all fully connected layers in the embeddings, encoder, and pooler.
|
||||
attention_probs_dropout_prob (:obj:`float`, `optional`, defaults to 0.1):
|
||||
The dropout ratio for the attention probabilities.
|
||||
max_position_embeddings (:obj:`int`, `optional`, defaults to 512):
|
||||
The maximum sequence length that this model might ever be used with. Typically set this to something large
|
||||
just in case (e.g., 512 or 1024 or 2048).
|
||||
type_vocab_size (:obj:`int`, `optional`, defaults to 2):
|
||||
The vocabulary size of the :obj:`token_type_ids` passed when calling :class:`~transformers.BertModel` or
|
||||
:class:`~transformers.TFBertModel`.
|
||||
initializer_range (:obj:`float`, `optional`, defaults to 0.02):
|
||||
The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
|
||||
layer_norm_eps (:obj:`float`, `optional`, defaults to 1e-12):
|
||||
The epsilon used by the layer normalization layers.
|
||||
position_embedding_type (:obj:`str`, `optional`, defaults to :obj:`"absolute"`):
|
||||
Type of position embedding. Choose one of :obj:`"absolute"`, :obj:`"relative_key"`,
|
||||
:obj:`"relative_key_query"`. For positional embeddings use :obj:`"absolute"`. For more information on
|
||||
:obj:`"relative_key"`, please refer to `Self-Attention with Relative Position Representations (Shaw et al.)
|
||||
<https://arxiv.org/abs/1803.02155>`__. For more information on :obj:`"relative_key_query"`, please refer to
|
||||
`Method 4` in `Improve Transformer Models with Better Relative Position Embeddings (Huang et al.)
|
||||
<https://arxiv.org/abs/2009.13658>`__.
|
||||
use_cache (:obj:`bool`, `optional`, defaults to :obj:`True`):
|
||||
Whether or not the model should return the last key/values attentions (not used by all models). Only
|
||||
relevant if ``config.is_decoder=True``.
|
||||
classifier_dropout (:obj:`float`, `optional`):
|
||||
The dropout ratio for the classification head.
|
||||
adv_grad_factor (:obj:`float`, `optional`): This factor will be multiplied by the KL loss grad and then
|
||||
the result will be added to the original embedding.
|
||||
More details please check:https://arxiv.org/abs/1908.04577
|
||||
The range of this value should between 1e-3~1e-7
|
||||
adv_bound (:obj:`float`, `optional`): adv_bound is used to cut the top and the bottom bound of
|
||||
the produced embedding.
|
||||
If not provided, 2 * sigma will be used as the adv_bound factor
|
||||
sigma (:obj:`float`, `optional`): The std factor used to produce a 0 mean normal distribution.
|
||||
If adv_bound not provided, 2 * sigma will be used as the adv_bound factor
|
||||
"""
|
||||
|
||||
model_type = 'plug-mental'
|
||||
|
||||
def __init__(self,
|
||||
vocab_size=30522,
|
||||
hidden_size=768,
|
||||
num_hidden_layers=12,
|
||||
num_attention_heads=12,
|
||||
intermediate_size=3072,
|
||||
hidden_act='gelu',
|
||||
hidden_dropout_prob=0.1,
|
||||
attention_probs_dropout_prob=0.1,
|
||||
max_position_embeddings=512,
|
||||
type_vocab_size=2,
|
||||
initializer_range=0.02,
|
||||
layer_norm_eps=1e-12,
|
||||
pad_token_id=0,
|
||||
position_embedding_type='absolute',
|
||||
use_cache=True,
|
||||
classifier_dropout=None,
|
||||
adapter_size=768,
|
||||
adapter_transformer_layers=1,
|
||||
adapter_list=[11],
|
||||
**kwargs):
|
||||
super().__init__(pad_token_id=pad_token_id, **kwargs)
|
||||
|
||||
self.vocab_size = vocab_size
|
||||
self.hidden_size = hidden_size
|
||||
self.num_hidden_layers = num_hidden_layers
|
||||
self.num_attention_heads = num_attention_heads
|
||||
self.hidden_act = hidden_act
|
||||
self.intermediate_size = intermediate_size
|
||||
self.hidden_dropout_prob = hidden_dropout_prob
|
||||
self.attention_probs_dropout_prob = attention_probs_dropout_prob
|
||||
self.max_position_embeddings = max_position_embeddings
|
||||
self.type_vocab_size = type_vocab_size
|
||||
self.initializer_range = initializer_range
|
||||
self.layer_norm_eps = layer_norm_eps
|
||||
self.position_embedding_type = position_embedding_type
|
||||
self.use_cache = use_cache
|
||||
self.classifier_dropout = classifier_dropout
|
||||
self.output_hidden_states = True
|
||||
# adv_grad_factor, used in adv loss.
|
||||
# Users can check adv_utils.py for details.
|
||||
# if adv_grad_factor set to None, no adv loss will not applied to the model.
|
||||
self.adv_grad_factor = None
|
||||
# sigma value, used in adv loss.
|
||||
self.sigma = 5e-6 if 'sigma' not in kwargs else kwargs['sigma']
|
||||
# adv_bound value, used in adv loss.
|
||||
self.adv_bound = 2 * self.sigma if 'adv_bound' not in kwargs else kwargs[
|
||||
'adv_bound']
|
||||
|
||||
# adapter config
|
||||
self.adapter_size = adapter_size
|
||||
self.adapter_transformer_layers = adapter_transformer_layers
|
||||
self.adapter_list = adapter_list
|
||||
223
modelscope/models/nlp/plug_mental/text_classification.py
Normal file
223
modelscope/models/nlp/plug_mental/text_classification.py
Normal file
@@ -0,0 +1,223 @@
|
||||
# Copyright 2021-2022 The Alibaba DAMO NLP Team Authors.
|
||||
# Copyright 2018 The Google AI Language Team Authors and The HuggingFace Inc. team.
|
||||
# Copyright (c) 2018, NVIDIA CORPORATION. All rights reserved.
|
||||
# All rights reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
import torch
|
||||
import torch.nn as nn
|
||||
import torch.utils.checkpoint
|
||||
from torch.nn import BCEWithLogitsLoss, CrossEntropyLoss, MSELoss
|
||||
|
||||
from modelscope.metainfo import Models
|
||||
from modelscope.models.builder import MODELS
|
||||
from modelscope.outputs import AttentionTextClassificationModelOutput
|
||||
from modelscope.utils import logger as logging
|
||||
from modelscope.utils.constant import Tasks
|
||||
from .adv_utils import compute_adv_loss
|
||||
from .backbone import PlugMentalModel, PlugMentalPreTrainedModel
|
||||
from .configuration import PlugMentalConfig
|
||||
|
||||
logger = logging.get_logger()
|
||||
|
||||
|
||||
@MODELS.register_module(
|
||||
Tasks.text_classification, module_name=Models.plug_mental)
|
||||
@MODELS.register_module(Tasks.nli, module_name=Models.plug_mental)
|
||||
@MODELS.register_module(
|
||||
Tasks.sentiment_classification, module_name=Models.plug_mental)
|
||||
@MODELS.register_module(
|
||||
Tasks.sentence_similarity, module_name=Models.plug_mental)
|
||||
@MODELS.register_module(
|
||||
Tasks.zero_shot_classification, module_name=Models.plug_mental)
|
||||
class PlugMentalForSequenceClassification(PlugMentalPreTrainedModel):
|
||||
r"""PlugMental Model transformer with a sequence classification/regression head on top
|
||||
(a linear layer on top of the pooled output) e.g. for GLUE tasks.
|
||||
|
||||
This model inherits from :class:`~transformers.PreTrainedModel`. Check the superclass documentation for the generic
|
||||
methods the library implements for all its model (such as downloading or saving, resizing the input embeddings,
|
||||
pruning heads etc.)
|
||||
|
||||
This model is also a PyTorch `torch.nn.Module <https://pytorch.org/docs/stable/nn.html#torch.nn.Module>`__
|
||||
subclass. Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to
|
||||
general usage and behavior.
|
||||
|
||||
Preprocessor:
|
||||
This is the text classification model of PlugMental, the preprocessor of this model
|
||||
is `modelscope.preprocessors.TextClassificationTransformersPreprocessor`.
|
||||
|
||||
Trainer:
|
||||
This model is a normal PyTorch model, and can be trained by variable trainers, like EpochBasedTrainer,
|
||||
NlpEpochBasedTrainer, or trainers from other frameworks.
|
||||
The preferred trainer in ModelScope is NlpEpochBasedTrainer.
|
||||
|
||||
Parameters:
|
||||
config (:class:`~modelscope.models.nlp.plug_mental.PlugMentalConfig`): Model configuration class with
|
||||
all the parameters of the model.
|
||||
Initializing with a config file does not load the weights associated with the model, only the
|
||||
configuration. Check out the :meth:`~transformers.PreTrainedModel.from_pretrained` method to load the model
|
||||
weights.
|
||||
"""
|
||||
|
||||
def __init__(self, config: PlugMentalConfig, **kwargs):
|
||||
super().__init__(config)
|
||||
self.num_labels = config.num_labels
|
||||
self.config = config
|
||||
if self.config.adv_grad_factor is None:
|
||||
logger.warning(
|
||||
'Adv parameters not set, skipping compute_adv_loss.')
|
||||
|
||||
PlugMentalForSequenceClassification.base_model_prefix = getattr(
|
||||
config, 'base_model_prefix',
|
||||
PlugMentalForSequenceClassification.base_model_prefix)
|
||||
setattr(self, self.base_model_prefix, PlugMentalModel(config))
|
||||
classifier_dropout = (
|
||||
config.classifier_dropout if config.classifier_dropout is not None
|
||||
else config.hidden_dropout_prob)
|
||||
self.dropout = nn.Dropout(classifier_dropout)
|
||||
self.classifier = nn.Linear(config.hidden_size, config.num_labels)
|
||||
self.init_weights()
|
||||
|
||||
def _forward_call(self, **kwargs):
|
||||
outputs = self.base_model(**kwargs)
|
||||
pooled_output = outputs[1]
|
||||
pooled_output = self.dropout(pooled_output)
|
||||
logits = self.classifier(pooled_output)
|
||||
outputs['logits'] = logits
|
||||
outputs.kwargs = kwargs
|
||||
return outputs
|
||||
|
||||
def forward(self,
|
||||
input_ids=None,
|
||||
attention_mask=None,
|
||||
token_type_ids=None,
|
||||
position_ids=None,
|
||||
head_mask=None,
|
||||
inputs_embeds=None,
|
||||
labels=None,
|
||||
output_attentions=None,
|
||||
output_hidden_states=None,
|
||||
return_dict=None,
|
||||
*args,
|
||||
**kwargs):
|
||||
r"""
|
||||
Args:
|
||||
input_ids (:obj:`torch.LongTensor` of shape :obj:`(batch_size, sequence_length)`):
|
||||
Indices of input sequence tokens in the vocabulary.
|
||||
|
||||
|
||||
attention_mask (:obj:`torch.FloatTensor` of shape :obj:`(batch_size, sequence_length)`, `optional`):
|
||||
Mask to avoid performing attention on padding token indices. Mask values selected in ``[0, 1]``:
|
||||
|
||||
- 1 for tokens that are **not masked**,
|
||||
- 0 for tokens that are **masked**.
|
||||
|
||||
token_type_ids (:obj:`torch.LongTensor` of shape :obj:`(batch_size, sequence_length)`, `optional`):
|
||||
Segment token indices to indicate first and second portions of the inputs. Indices are selected in ``[0,
|
||||
1]``:
|
||||
|
||||
- 0 corresponds to a `sentence A` token,
|
||||
- 1 corresponds to a `sentence B` token.
|
||||
|
||||
position_ids (:obj:`torch.LongTensor` of shape :obj:`(batch_size, sequence_length)`, `optional`):
|
||||
Indices of positions of each input sequence tokens in the position embeddings. Selected in the range
|
||||
``[0, config.max_position_embeddings - 1]``.
|
||||
|
||||
head_mask (:obj:`torch.FloatTensor` of shape :obj:`(num_heads,)` or :obj:`(num_layers, num_heads)`,
|
||||
`optional`):
|
||||
Mask to nullify selected heads of the self-attention modules. Mask values selected in ``[0, 1]``:
|
||||
|
||||
- 1 indicates the head is **not masked**,
|
||||
- 0 indicates the head is **masked**.
|
||||
|
||||
inputs_embeds (:obj:`torch.FloatTensor` of shape :obj:`(batch_size, sequence_length, hidden_size)`,
|
||||
`optional`):
|
||||
Optionally, instead of passing :obj:`input_ids` you can choose to directly pass an embedded
|
||||
representation. This is useful if you want more control over how to convert :obj:`input_ids`
|
||||
indices into associated vectors than the model's internal embedding lookup matrix.
|
||||
output_attentions (:obj:`bool`, `optional`):
|
||||
Whether or not to return the attentions tensors of all attention layers. See
|
||||
``attentions`` under returned tensors for more detail.
|
||||
output_hidden_states (:obj:`bool`, `optional`):
|
||||
Whether or not to return the hidden states of all layers. See ``hidden_states`` under
|
||||
returned tensors for more detail.
|
||||
return_dict (:obj:`bool`, `optional`):
|
||||
Whether or not to return a :class:`~transformers.ModelOutput` instead of a plain tuple.
|
||||
labels (:obj:`torch.LongTensor` of shape :obj:`(batch_size,)`, `optional`):
|
||||
Labels for computing the sequence classification/regression loss. Indices should be in :obj:`[0, ...,
|
||||
config.num_labels - 1]`. If :obj:`config.num_labels == 1` a regression loss is computed
|
||||
(Mean-Square loss), If :obj:`config.num_labels > 1` a classification loss is computed (Cross-Entropy).
|
||||
|
||||
Returns:
|
||||
Returns `modelscope.outputs.AttentionTextClassificationModelOutput`
|
||||
"""
|
||||
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
|
||||
if not return_dict:
|
||||
logger.error('Return tuple in sbert is not supported now.')
|
||||
outputs = self._forward_call(
|
||||
input_ids=input_ids,
|
||||
attention_mask=attention_mask,
|
||||
token_type_ids=token_type_ids,
|
||||
position_ids=position_ids,
|
||||
head_mask=head_mask,
|
||||
inputs_embeds=inputs_embeds,
|
||||
output_attentions=output_attentions,
|
||||
output_hidden_states=output_hidden_states,
|
||||
return_dict=return_dict)
|
||||
return self.compute_loss(outputs, labels, **outputs.kwargs)
|
||||
|
||||
def compute_loss(self, outputs, labels, **kwargs):
|
||||
logits = outputs.logits
|
||||
embedding_output = outputs.embedding_output
|
||||
loss = None
|
||||
if labels is not None:
|
||||
if self.config.problem_type is None:
|
||||
if self.num_labels == 1:
|
||||
self.config.problem_type = 'regression'
|
||||
elif self.num_labels > 1 and (labels.dtype == torch.long
|
||||
or labels.dtype == torch.int):
|
||||
self.config.problem_type = 'single_label_classification'
|
||||
else:
|
||||
self.config.problem_type = 'multi_label_classification'
|
||||
|
||||
if self.config.problem_type == 'regression':
|
||||
loss_fct = MSELoss()
|
||||
if self.num_labels == 1:
|
||||
loss = loss_fct(logits.squeeze(), labels.squeeze())
|
||||
else:
|
||||
loss = loss_fct(logits, labels)
|
||||
elif self.config.problem_type == 'single_label_classification':
|
||||
loss_fct = CrossEntropyLoss()
|
||||
loss = loss_fct(
|
||||
logits.view(-1, self.num_labels), labels.view(-1))
|
||||
if self.config.adv_grad_factor is not None and self.training:
|
||||
loss = compute_adv_loss(
|
||||
embedding=embedding_output,
|
||||
model=self._forward_call,
|
||||
ori_logits=logits,
|
||||
ori_loss=loss,
|
||||
adv_bound=self.config.adv_bound,
|
||||
adv_grad_factor=self.config.adv_grad_factor,
|
||||
sigma=self.config.sigma,
|
||||
**kwargs)
|
||||
elif self.config.problem_type == 'multi_label_classification':
|
||||
loss_fct = BCEWithLogitsLoss()
|
||||
loss = loss_fct(logits, labels)
|
||||
|
||||
return AttentionTextClassificationModelOutput(
|
||||
loss=loss,
|
||||
logits=logits,
|
||||
hidden_states=outputs.hidden_states,
|
||||
attentions=outputs.attentions,
|
||||
)
|
||||
@@ -82,7 +82,8 @@ class NLPTokenizer:
|
||||
model_dir) if model_dir is not None else tokenizer()
|
||||
|
||||
if model_type in (Models.structbert, Models.gpt3, Models.palm,
|
||||
Models.plug, Models.megatron_bert):
|
||||
Models.plug, Models.megatron_bert,
|
||||
Models.plug_mental):
|
||||
from transformers import BertTokenizer, BertTokenizerFast
|
||||
tokenizer = BertTokenizerFast if self.use_fast else BertTokenizer
|
||||
return tokenizer.from_pretrained(
|
||||
|
||||
108
tests/trainers/test_finetune_plug_mental.py
Normal file
108
tests/trainers/test_finetune_plug_mental.py
Normal file
@@ -0,0 +1,108 @@
|
||||
# Copyright (c) Alibaba, Inc. and its affiliates.
|
||||
import os
|
||||
import shutil
|
||||
import tempfile
|
||||
import unittest
|
||||
from typing import Any, Callable, Dict, List, NewType, Optional, Tuple, Union
|
||||
|
||||
import torch
|
||||
from transformers.tokenization_utils_base import PreTrainedTokenizerBase
|
||||
|
||||
from modelscope.metainfo import Preprocessors, Trainers
|
||||
from modelscope.models import Model
|
||||
from modelscope.msdatasets import MsDataset
|
||||
from modelscope.pipelines import pipeline
|
||||
from modelscope.trainers import build_trainer
|
||||
from modelscope.utils.constant import ModelFile, Tasks
|
||||
from modelscope.utils.test_utils import test_level
|
||||
|
||||
|
||||
class TestFinetunePlugMental(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
print(('Testing %s.%s' % (type(self).__name__, self._testMethodName)))
|
||||
self.tmp_dir = tempfile.TemporaryDirectory().name
|
||||
if not os.path.exists(self.tmp_dir):
|
||||
os.makedirs(self.tmp_dir)
|
||||
|
||||
def tearDown(self):
|
||||
shutil.rmtree(self.tmp_dir)
|
||||
super().tearDown()
|
||||
|
||||
def finetune(self,
|
||||
model_id,
|
||||
train_dataset,
|
||||
eval_dataset,
|
||||
name=Trainers.nlp_base_trainer,
|
||||
cfg_modify_fn=None,
|
||||
**kwargs):
|
||||
kwargs = dict(
|
||||
model=model_id,
|
||||
train_dataset=train_dataset,
|
||||
eval_dataset=eval_dataset,
|
||||
work_dir=self.tmp_dir,
|
||||
cfg_modify_fn=cfg_modify_fn,
|
||||
**kwargs)
|
||||
|
||||
os.environ['LOCAL_RANK'] = '0'
|
||||
trainer = build_trainer(name=name, default_args=kwargs)
|
||||
trainer.train()
|
||||
results_files = os.listdir(self.tmp_dir)
|
||||
self.assertIn(f'{trainer.timestamp}.log.json', results_files)
|
||||
for i in range(self.epoch_num):
|
||||
self.assertIn(f'epoch_{i + 1}.pth', results_files)
|
||||
|
||||
output_files = os.listdir(
|
||||
os.path.join(self.tmp_dir, ModelFile.TRAIN_OUTPUT_DIR))
|
||||
self.assertIn(ModelFile.CONFIGURATION, output_files)
|
||||
self.assertIn(ModelFile.TORCH_MODEL_BIN_FILE, output_files)
|
||||
copy_src_files = os.listdir(trainer.model_dir)
|
||||
|
||||
print(f'copy_src_files are {copy_src_files}')
|
||||
print(f'output_files are {output_files}')
|
||||
for item in copy_src_files:
|
||||
if not item.startswith('.'):
|
||||
self.assertIn(item, output_files)
|
||||
|
||||
def pipeline_sentence_similarity(self, model_dir):
|
||||
sentence1 = '今天气温比昨天高么?'
|
||||
sentence2 = '今天湿度比昨天高么?'
|
||||
model = Model.from_pretrained(model_dir)
|
||||
pipeline_ins = pipeline(task=Tasks.sentence_similarity, model=model)
|
||||
print(pipeline_ins(input=(sentence1, sentence2)))
|
||||
|
||||
@unittest.skip
|
||||
def test_finetune_afqmc(self):
|
||||
"""This unittest is used to reproduce the clue:afqmc dataset + plug meantal model training results.
|
||||
|
||||
User can train a custom dataset by modifying this piece of code and comment the @unittest.skip.
|
||||
"""
|
||||
|
||||
def cfg_modify_fn(cfg):
|
||||
cfg.task = Tasks.sentence_similarity
|
||||
cfg['preprocessor'] = {'type': Preprocessors.sen_sim_tokenizer}
|
||||
cfg.train.optimizer.lr = 2e-5
|
||||
cfg['dataset'] = {
|
||||
'train': {
|
||||
'labels': ['0', '1'],
|
||||
'first_sequence': 'sentence1',
|
||||
'second_sequence': 'sentence2',
|
||||
'label': 'label',
|
||||
}
|
||||
}
|
||||
cfg.train.lr_scheduler.total_iters = int(
|
||||
len(dataset['train']) / 32) * cfg.train.max_epochs
|
||||
return cfg
|
||||
|
||||
dataset = MsDataset.load('clue', subset_name='afqmc')
|
||||
self.finetune(
|
||||
model_id='damo/nlp_plug-mental_backbone_base',
|
||||
train_dataset=dataset['train'],
|
||||
eval_dataset=dataset['validation'],
|
||||
cfg_modify_fn=cfg_modify_fn)
|
||||
output_dir = os.path.join(self.tmp_dir, ModelFile.TRAIN_OUTPUT_DIR)
|
||||
self.pipeline_sentence_similarity(output_dir)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user