add plug mental model

Link: https://code.alibaba-inc.com/Ali-MaaS/MaaS-lib/codereview/11549696

* add plug mental model code

* add test pipeline and fix annotation format bugs
This commit is contained in:
dawei.fdw
2023-02-06 10:57:20 +00:00
committed by wenmeng.zwm
parent e54694690f
commit 310e9c7dbf
9 changed files with 1768 additions and 2 deletions

View File

@@ -123,6 +123,7 @@ class Models(object):
unite = 'unite'
megatron_bert = 'megatron-bert'
use = 'user-satisfaction-estimation'
plug_mental = 'plug-mental'
# audio models
sambert_hifigan = 'sambert-hifigan'

View File

@@ -61,6 +61,8 @@ if TYPE_CHECKING:
from .bloom import BloomModel
from .unite import UniTEModel
from .use import UserSatisfactionEstimation
from .plug_mental import (PlugMentalConfig, PlugMentalModel,
PlugMentalForSequenceClassification)
else:
_import_structure = {
'backbones': ['SbertModel'],
@@ -127,7 +129,12 @@ else:
'gpt_neo': ['GPTNeoModel'],
'bloom': ['BloomModel'],
'unite': ['UniTEModel'],
'use': ['UserSatisfactionEstimation']
'use': ['UserSatisfactionEstimation'],
'plug_mental': [
'PlugMentalConfig',
'PlugMentalModel',
'PlugMentalForSequenceClassification',
]
}
import sys

View File

@@ -0,0 +1,39 @@
# Copyright 2021-2022 The Alibaba DAMO NLP Team Authors.
# All rights reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
from typing import TYPE_CHECKING
from modelscope.utils.import_utils import LazyImportModule
if TYPE_CHECKING:
from .backbone import (PlugMentalModel, PlugMentalPreTrainedModel)
from .configuration import PlugMentalConfig
from .text_classification import PlugMentalForSequenceClassification
else:
_import_structure = {
'backbone': ['PlugMentalModel', 'PlugMentalPreTrainedModel'],
'configuration': ['PlugMentalConfig'],
'text_classification': ['PlugMentalForSequenceClassification'],
}
import sys
sys.modules[__name__] = LazyImportModule(
__name__,
globals()['__file__'],
_import_structure,
module_spec=__spec__,
extra_objects={},
)

View File

@@ -0,0 +1,168 @@
# Copyright 2021-2022 The Alibaba DAMO NLP Team Authors.
# All rights reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import torch
from torch import nn
from modelscope.utils.logger import get_logger
logger = get_logger()
def _symmetric_kl_div(logits1, logits2, attention_mask=None):
"""
Calclate two logits' the KL div value symmetrically.
:param logits1: The first logit.
:param logits2: The second logit.
:param attention_mask: An optional attention_mask which is used to mask some element out.
This is usually useful in token_classification tasks.
If the shape of logits is [N1, N2, ... Nn, D], the shape of attention_mask should be [N1, N2, ... Nn]
:return: The mean loss.
"""
labels_num = logits1.shape[-1]
KLDiv = nn.KLDivLoss(reduction='none')
loss = torch.sum(
KLDiv(nn.LogSoftmax(dim=-1)(logits1),
nn.Softmax(dim=-1)(logits2)),
dim=-1) + torch.sum(
KLDiv(nn.LogSoftmax(dim=-1)(logits2),
nn.Softmax(dim=-1)(logits1)),
dim=-1)
if attention_mask is not None:
loss = torch.sum(
loss * attention_mask) / torch.sum(attention_mask) / labels_num
else:
loss = torch.mean(loss) / labels_num
return loss
def compute_adv_loss(embedding,
model,
ori_logits,
ori_loss,
adv_grad_factor,
adv_bound=None,
sigma=5e-6,
**kwargs):
"""
Calculate the adv loss of the model.
:param embedding: Original sentense embedding
:param model: The model, or the forward function(including decoder/classifier),
accept kwargs as input, output logits
:param ori_logits: The original logits outputed from the model function
:param ori_loss: The original loss
:param adv_grad_factor: This factor will be multipled by the KL loss grad and then the result will be added to
the original embedding.
More details please check:https://arxiv.org/abs/1908.04577
The range of this value always be 1e-3~1e-7
:param adv_bound: adv_bound is used to cut the top and the bottom bound of the produced embedding.
If not proveded, 2 * sigma will be used as the adv_bound factor
:param sigma: The std factor used to produce a 0 mean normal distribution.
If adv_bound not proveded, 2 * sigma will be used as the adv_bound factor
:param kwargs: the input param used in model function
:return: The original loss adds the adv loss
"""
adv_bound = adv_bound if adv_bound is not None else 2 * sigma
embedding_1 = embedding + embedding.data.new(embedding.size()).normal_(
0, sigma) # 95% in +- 1e-5
kwargs.pop('input_ids')
if 'inputs_embeds' in kwargs:
kwargs.pop('inputs_embeds')
with_attention_mask = False if 'with_attention_mask' not in kwargs else kwargs[
'with_attention_mask']
attention_mask = kwargs['attention_mask']
if not with_attention_mask:
attention_mask = None
if 'with_attention_mask' in kwargs:
kwargs.pop('with_attention_mask')
outputs = model(**kwargs, inputs_embeds=embedding_1)
v1_logits = outputs.logits
loss = _symmetric_kl_div(ori_logits, v1_logits, attention_mask)
emb_grad = torch.autograd.grad(loss, embedding_1)[0].data
emb_grad_norm = emb_grad.norm(
dim=2, keepdim=True, p=float('inf')).max(
1, keepdim=True)[0]
is_nan = torch.any(torch.isnan(emb_grad_norm))
if is_nan:
logger.warning('Nan occured when calculating adv loss.')
return ori_loss
emb_grad = emb_grad / (emb_grad_norm + 1e-6)
embedding_2 = embedding_1 + adv_grad_factor * emb_grad
embedding_2 = torch.max(embedding_1 - adv_bound, embedding_2)
embedding_2 = torch.min(embedding_1 + adv_bound, embedding_2)
outputs = model(**kwargs, inputs_embeds=embedding_2)
adv_logits = outputs.logits
adv_loss = _symmetric_kl_div(ori_logits, adv_logits, attention_mask)
return ori_loss + adv_loss
def compute_adv_loss_pair(embedding,
model,
start_logits,
end_logits,
ori_loss,
adv_grad_factor,
adv_bound=None,
sigma=5e-6,
**kwargs):
"""
Calculate the adv loss of the model. This function is used in the pair logits scenerio.
:param embedding: Original sentense embedding
:param model: The model, or the forward function(including decoder/classifier),
accept kwargs as input, output logits
:param start_logits: The original start logits outputed from the model function
:param end_logits: The original end logits outputed from the model function
:param ori_loss: The original loss
:param adv_grad_factor: This factor will be multipled by the KL loss grad and then the result will be added to
the original embedding.
More details please check:https://arxiv.org/abs/1908.04577
The range of this value always be 1e-3~1e-7
:param adv_bound: adv_bound is used to cut the top and the bottom bound of the produced embedding.
If not proveded, 2 * sigma will be used as the adv_bound factor
:param sigma: The std factor used to produce a 0 mean normal distribution.
If adv_bound not proveded, 2 * sigma will be used as the adv_bound factor
:param kwargs: the input param used in model function
:return: The original loss adds the adv loss
"""
adv_bound = adv_bound if adv_bound is not None else 2 * sigma
embedding_1 = embedding + embedding.data.new(embedding.size()).normal_(
0, sigma) # 95% in +- 1e-5
kwargs.pop('input_ids')
if 'inputs_embeds' in kwargs:
kwargs.pop('inputs_embeds')
outputs = model(**kwargs, inputs_embeds=embedding_1)
v1_logits_start, v1_logits_end = outputs.logits
loss = _symmetric_kl_div(start_logits,
v1_logits_start) + _symmetric_kl_div(
end_logits, v1_logits_end)
loss = loss / 2
emb_grad = torch.autograd.grad(loss, embedding_1)[0].data
emb_grad_norm = emb_grad.norm(
dim=2, keepdim=True, p=float('inf')).max(
1, keepdim=True)[0]
is_nan = torch.any(torch.isnan(emb_grad_norm))
if is_nan:
logger.warning('Nan occured when calculating pair adv loss.')
return ori_loss
emb_grad = emb_grad / emb_grad_norm
embedding_2 = embedding_1 + adv_grad_factor * emb_grad
embedding_2 = torch.max(embedding_1 - adv_bound, embedding_2)
embedding_2 = torch.min(embedding_1 + adv_bound, embedding_2)
outputs = model(**kwargs, inputs_embeds=embedding_2)
adv_logits_start, adv_logits_end = outputs.logits
adv_loss = _symmetric_kl_div(start_logits,
adv_logits_start) + _symmetric_kl_div(
end_logits, adv_logits_end)
return ori_loss + adv_loss

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,142 @@
# Copyright 2021-2022 The Alibaba DAMO NLP Team Authors.
# Copyright 2018 The Google AI Language Team Authors and The HuggingFace Inc. team.
# Copyright (c) 2018, NVIDIA CORPORATION. All rights reserved.
# All rights reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
""" PLUG mental model configuration, mainly copied from :class:`~transformers.BertConfig` """
from transformers import PretrainedConfig
from modelscope.utils import logger as logging
logger = logging.get_logger()
class PlugMentalConfig(PretrainedConfig):
r"""
This is the configuration class to store the configuration
of a :class:`~modelscope.models.nlp.plug_mental.PLugMentalModel`.
It is used to instantiate a PlugMental model according to the specified arguments.
Configuration objects inherit from :class:`~transformers.PretrainedConfig` and can be used to control the model
outputs. Read the documentation from :class:`~transformers.PretrainedConfig` for more information.
Args:
vocab_size (:obj:`int`, `optional`, defaults to 30522):
Vocabulary size of the BERT model. Defines the number of different tokens that can be represented by the
:obj:`inputs_ids` passed when calling :class:`~transformers.BertModel` or
:class:`~transformers.TFBertModel`.
hidden_size (:obj:`int`, `optional`, defaults to 768):
Dimensionality of the encoder layers and the pooler layer.
num_hidden_layers (:obj:`int`, `optional`, defaults to 12):
Number of hidden layers in the Transformer encoder.
num_attention_heads (:obj:`int`, `optional`, defaults to 12):
Number of attention heads for each attention layer in the Transformer encoder.
intermediate_size (:obj:`int`, `optional`, defaults to 3072):
Dimensionality of the "intermediate" (often named feed-forward) layer in the Transformer encoder.
hidden_act (:obj:`str` or :obj:`Callable`, `optional`, defaults to :obj:`"gelu"`):
The non-linear activation function (function or string) in the encoder and pooler. If string,
:obj:`"gelu"`, :obj:`"relu"`, :obj:`"silu"` and :obj:`"gelu_new"` are supported.
hidden_dropout_prob (:obj:`float`, `optional`, defaults to 0.1):
The dropout probability for all fully connected layers in the embeddings, encoder, and pooler.
attention_probs_dropout_prob (:obj:`float`, `optional`, defaults to 0.1):
The dropout ratio for the attention probabilities.
max_position_embeddings (:obj:`int`, `optional`, defaults to 512):
The maximum sequence length that this model might ever be used with. Typically set this to something large
just in case (e.g., 512 or 1024 or 2048).
type_vocab_size (:obj:`int`, `optional`, defaults to 2):
The vocabulary size of the :obj:`token_type_ids` passed when calling :class:`~transformers.BertModel` or
:class:`~transformers.TFBertModel`.
initializer_range (:obj:`float`, `optional`, defaults to 0.02):
The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
layer_norm_eps (:obj:`float`, `optional`, defaults to 1e-12):
The epsilon used by the layer normalization layers.
position_embedding_type (:obj:`str`, `optional`, defaults to :obj:`"absolute"`):
Type of position embedding. Choose one of :obj:`"absolute"`, :obj:`"relative_key"`,
:obj:`"relative_key_query"`. For positional embeddings use :obj:`"absolute"`. For more information on
:obj:`"relative_key"`, please refer to `Self-Attention with Relative Position Representations (Shaw et al.)
<https://arxiv.org/abs/1803.02155>`__. For more information on :obj:`"relative_key_query"`, please refer to
`Method 4` in `Improve Transformer Models with Better Relative Position Embeddings (Huang et al.)
<https://arxiv.org/abs/2009.13658>`__.
use_cache (:obj:`bool`, `optional`, defaults to :obj:`True`):
Whether or not the model should return the last key/values attentions (not used by all models). Only
relevant if ``config.is_decoder=True``.
classifier_dropout (:obj:`float`, `optional`):
The dropout ratio for the classification head.
adv_grad_factor (:obj:`float`, `optional`): This factor will be multiplied by the KL loss grad and then
the result will be added to the original embedding.
More details please check:https://arxiv.org/abs/1908.04577
The range of this value should between 1e-3~1e-7
adv_bound (:obj:`float`, `optional`): adv_bound is used to cut the top and the bottom bound of
the produced embedding.
If not provided, 2 * sigma will be used as the adv_bound factor
sigma (:obj:`float`, `optional`): The std factor used to produce a 0 mean normal distribution.
If adv_bound not provided, 2 * sigma will be used as the adv_bound factor
"""
model_type = 'plug-mental'
def __init__(self,
vocab_size=30522,
hidden_size=768,
num_hidden_layers=12,
num_attention_heads=12,
intermediate_size=3072,
hidden_act='gelu',
hidden_dropout_prob=0.1,
attention_probs_dropout_prob=0.1,
max_position_embeddings=512,
type_vocab_size=2,
initializer_range=0.02,
layer_norm_eps=1e-12,
pad_token_id=0,
position_embedding_type='absolute',
use_cache=True,
classifier_dropout=None,
adapter_size=768,
adapter_transformer_layers=1,
adapter_list=[11],
**kwargs):
super().__init__(pad_token_id=pad_token_id, **kwargs)
self.vocab_size = vocab_size
self.hidden_size = hidden_size
self.num_hidden_layers = num_hidden_layers
self.num_attention_heads = num_attention_heads
self.hidden_act = hidden_act
self.intermediate_size = intermediate_size
self.hidden_dropout_prob = hidden_dropout_prob
self.attention_probs_dropout_prob = attention_probs_dropout_prob
self.max_position_embeddings = max_position_embeddings
self.type_vocab_size = type_vocab_size
self.initializer_range = initializer_range
self.layer_norm_eps = layer_norm_eps
self.position_embedding_type = position_embedding_type
self.use_cache = use_cache
self.classifier_dropout = classifier_dropout
self.output_hidden_states = True
# adv_grad_factor, used in adv loss.
# Users can check adv_utils.py for details.
# if adv_grad_factor set to None, no adv loss will not applied to the model.
self.adv_grad_factor = None
# sigma value, used in adv loss.
self.sigma = 5e-6 if 'sigma' not in kwargs else kwargs['sigma']
# adv_bound value, used in adv loss.
self.adv_bound = 2 * self.sigma if 'adv_bound' not in kwargs else kwargs[
'adv_bound']
# adapter config
self.adapter_size = adapter_size
self.adapter_transformer_layers = adapter_transformer_layers
self.adapter_list = adapter_list

View File

@@ -0,0 +1,223 @@
# Copyright 2021-2022 The Alibaba DAMO NLP Team Authors.
# Copyright 2018 The Google AI Language Team Authors and The HuggingFace Inc. team.
# Copyright (c) 2018, NVIDIA CORPORATION. All rights reserved.
# All rights reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import torch
import torch.nn as nn
import torch.utils.checkpoint
from torch.nn import BCEWithLogitsLoss, CrossEntropyLoss, MSELoss
from modelscope.metainfo import Models
from modelscope.models.builder import MODELS
from modelscope.outputs import AttentionTextClassificationModelOutput
from modelscope.utils import logger as logging
from modelscope.utils.constant import Tasks
from .adv_utils import compute_adv_loss
from .backbone import PlugMentalModel, PlugMentalPreTrainedModel
from .configuration import PlugMentalConfig
logger = logging.get_logger()
@MODELS.register_module(
Tasks.text_classification, module_name=Models.plug_mental)
@MODELS.register_module(Tasks.nli, module_name=Models.plug_mental)
@MODELS.register_module(
Tasks.sentiment_classification, module_name=Models.plug_mental)
@MODELS.register_module(
Tasks.sentence_similarity, module_name=Models.plug_mental)
@MODELS.register_module(
Tasks.zero_shot_classification, module_name=Models.plug_mental)
class PlugMentalForSequenceClassification(PlugMentalPreTrainedModel):
r"""PlugMental Model transformer with a sequence classification/regression head on top
(a linear layer on top of the pooled output) e.g. for GLUE tasks.
This model inherits from :class:`~transformers.PreTrainedModel`. Check the superclass documentation for the generic
methods the library implements for all its model (such as downloading or saving, resizing the input embeddings,
pruning heads etc.)
This model is also a PyTorch `torch.nn.Module <https://pytorch.org/docs/stable/nn.html#torch.nn.Module>`__
subclass. Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to
general usage and behavior.
Preprocessor:
This is the text classification model of PlugMental, the preprocessor of this model
is `modelscope.preprocessors.TextClassificationTransformersPreprocessor`.
Trainer:
This model is a normal PyTorch model, and can be trained by variable trainers, like EpochBasedTrainer,
NlpEpochBasedTrainer, or trainers from other frameworks.
The preferred trainer in ModelScope is NlpEpochBasedTrainer.
Parameters:
config (:class:`~modelscope.models.nlp.plug_mental.PlugMentalConfig`): Model configuration class with
all the parameters of the model.
Initializing with a config file does not load the weights associated with the model, only the
configuration. Check out the :meth:`~transformers.PreTrainedModel.from_pretrained` method to load the model
weights.
"""
def __init__(self, config: PlugMentalConfig, **kwargs):
super().__init__(config)
self.num_labels = config.num_labels
self.config = config
if self.config.adv_grad_factor is None:
logger.warning(
'Adv parameters not set, skipping compute_adv_loss.')
PlugMentalForSequenceClassification.base_model_prefix = getattr(
config, 'base_model_prefix',
PlugMentalForSequenceClassification.base_model_prefix)
setattr(self, self.base_model_prefix, PlugMentalModel(config))
classifier_dropout = (
config.classifier_dropout if config.classifier_dropout is not None
else config.hidden_dropout_prob)
self.dropout = nn.Dropout(classifier_dropout)
self.classifier = nn.Linear(config.hidden_size, config.num_labels)
self.init_weights()
def _forward_call(self, **kwargs):
outputs = self.base_model(**kwargs)
pooled_output = outputs[1]
pooled_output = self.dropout(pooled_output)
logits = self.classifier(pooled_output)
outputs['logits'] = logits
outputs.kwargs = kwargs
return outputs
def forward(self,
input_ids=None,
attention_mask=None,
token_type_ids=None,
position_ids=None,
head_mask=None,
inputs_embeds=None,
labels=None,
output_attentions=None,
output_hidden_states=None,
return_dict=None,
*args,
**kwargs):
r"""
Args:
input_ids (:obj:`torch.LongTensor` of shape :obj:`(batch_size, sequence_length)`):
Indices of input sequence tokens in the vocabulary.
attention_mask (:obj:`torch.FloatTensor` of shape :obj:`(batch_size, sequence_length)`, `optional`):
Mask to avoid performing attention on padding token indices. Mask values selected in ``[0, 1]``:
- 1 for tokens that are **not masked**,
- 0 for tokens that are **masked**.
token_type_ids (:obj:`torch.LongTensor` of shape :obj:`(batch_size, sequence_length)`, `optional`):
Segment token indices to indicate first and second portions of the inputs. Indices are selected in ``[0,
1]``:
- 0 corresponds to a `sentence A` token,
- 1 corresponds to a `sentence B` token.
position_ids (:obj:`torch.LongTensor` of shape :obj:`(batch_size, sequence_length)`, `optional`):
Indices of positions of each input sequence tokens in the position embeddings. Selected in the range
``[0, config.max_position_embeddings - 1]``.
head_mask (:obj:`torch.FloatTensor` of shape :obj:`(num_heads,)` or :obj:`(num_layers, num_heads)`,
`optional`):
Mask to nullify selected heads of the self-attention modules. Mask values selected in ``[0, 1]``:
- 1 indicates the head is **not masked**,
- 0 indicates the head is **masked**.
inputs_embeds (:obj:`torch.FloatTensor` of shape :obj:`(batch_size, sequence_length, hidden_size)`,
`optional`):
Optionally, instead of passing :obj:`input_ids` you can choose to directly pass an embedded
representation. This is useful if you want more control over how to convert :obj:`input_ids`
indices into associated vectors than the model's internal embedding lookup matrix.
output_attentions (:obj:`bool`, `optional`):
Whether or not to return the attentions tensors of all attention layers. See
``attentions`` under returned tensors for more detail.
output_hidden_states (:obj:`bool`, `optional`):
Whether or not to return the hidden states of all layers. See ``hidden_states`` under
returned tensors for more detail.
return_dict (:obj:`bool`, `optional`):
Whether or not to return a :class:`~transformers.ModelOutput` instead of a plain tuple.
labels (:obj:`torch.LongTensor` of shape :obj:`(batch_size,)`, `optional`):
Labels for computing the sequence classification/regression loss. Indices should be in :obj:`[0, ...,
config.num_labels - 1]`. If :obj:`config.num_labels == 1` a regression loss is computed
(Mean-Square loss), If :obj:`config.num_labels > 1` a classification loss is computed (Cross-Entropy).
Returns:
Returns `modelscope.outputs.AttentionTextClassificationModelOutput`
"""
return_dict = return_dict if return_dict is not None else self.config.use_return_dict
if not return_dict:
logger.error('Return tuple in sbert is not supported now.')
outputs = self._forward_call(
input_ids=input_ids,
attention_mask=attention_mask,
token_type_ids=token_type_ids,
position_ids=position_ids,
head_mask=head_mask,
inputs_embeds=inputs_embeds,
output_attentions=output_attentions,
output_hidden_states=output_hidden_states,
return_dict=return_dict)
return self.compute_loss(outputs, labels, **outputs.kwargs)
def compute_loss(self, outputs, labels, **kwargs):
logits = outputs.logits
embedding_output = outputs.embedding_output
loss = None
if labels is not None:
if self.config.problem_type is None:
if self.num_labels == 1:
self.config.problem_type = 'regression'
elif self.num_labels > 1 and (labels.dtype == torch.long
or labels.dtype == torch.int):
self.config.problem_type = 'single_label_classification'
else:
self.config.problem_type = 'multi_label_classification'
if self.config.problem_type == 'regression':
loss_fct = MSELoss()
if self.num_labels == 1:
loss = loss_fct(logits.squeeze(), labels.squeeze())
else:
loss = loss_fct(logits, labels)
elif self.config.problem_type == 'single_label_classification':
loss_fct = CrossEntropyLoss()
loss = loss_fct(
logits.view(-1, self.num_labels), labels.view(-1))
if self.config.adv_grad_factor is not None and self.training:
loss = compute_adv_loss(
embedding=embedding_output,
model=self._forward_call,
ori_logits=logits,
ori_loss=loss,
adv_bound=self.config.adv_bound,
adv_grad_factor=self.config.adv_grad_factor,
sigma=self.config.sigma,
**kwargs)
elif self.config.problem_type == 'multi_label_classification':
loss_fct = BCEWithLogitsLoss()
loss = loss_fct(logits, labels)
return AttentionTextClassificationModelOutput(
loss=loss,
logits=logits,
hidden_states=outputs.hidden_states,
attentions=outputs.attentions,
)

View File

@@ -82,7 +82,8 @@ class NLPTokenizer:
model_dir) if model_dir is not None else tokenizer()
if model_type in (Models.structbert, Models.gpt3, Models.palm,
Models.plug, Models.megatron_bert):
Models.plug, Models.megatron_bert,
Models.plug_mental):
from transformers import BertTokenizer, BertTokenizerFast
tokenizer = BertTokenizerFast if self.use_fast else BertTokenizer
return tokenizer.from_pretrained(

View File

@@ -0,0 +1,108 @@
# Copyright (c) Alibaba, Inc. and its affiliates.
import os
import shutil
import tempfile
import unittest
from typing import Any, Callable, Dict, List, NewType, Optional, Tuple, Union
import torch
from transformers.tokenization_utils_base import PreTrainedTokenizerBase
from modelscope.metainfo import Preprocessors, Trainers
from modelscope.models import Model
from modelscope.msdatasets import MsDataset
from modelscope.pipelines import pipeline
from modelscope.trainers import build_trainer
from modelscope.utils.constant import ModelFile, Tasks
from modelscope.utils.test_utils import test_level
class TestFinetunePlugMental(unittest.TestCase):
def setUp(self):
print(('Testing %s.%s' % (type(self).__name__, self._testMethodName)))
self.tmp_dir = tempfile.TemporaryDirectory().name
if not os.path.exists(self.tmp_dir):
os.makedirs(self.tmp_dir)
def tearDown(self):
shutil.rmtree(self.tmp_dir)
super().tearDown()
def finetune(self,
model_id,
train_dataset,
eval_dataset,
name=Trainers.nlp_base_trainer,
cfg_modify_fn=None,
**kwargs):
kwargs = dict(
model=model_id,
train_dataset=train_dataset,
eval_dataset=eval_dataset,
work_dir=self.tmp_dir,
cfg_modify_fn=cfg_modify_fn,
**kwargs)
os.environ['LOCAL_RANK'] = '0'
trainer = build_trainer(name=name, default_args=kwargs)
trainer.train()
results_files = os.listdir(self.tmp_dir)
self.assertIn(f'{trainer.timestamp}.log.json', results_files)
for i in range(self.epoch_num):
self.assertIn(f'epoch_{i + 1}.pth', results_files)
output_files = os.listdir(
os.path.join(self.tmp_dir, ModelFile.TRAIN_OUTPUT_DIR))
self.assertIn(ModelFile.CONFIGURATION, output_files)
self.assertIn(ModelFile.TORCH_MODEL_BIN_FILE, output_files)
copy_src_files = os.listdir(trainer.model_dir)
print(f'copy_src_files are {copy_src_files}')
print(f'output_files are {output_files}')
for item in copy_src_files:
if not item.startswith('.'):
self.assertIn(item, output_files)
def pipeline_sentence_similarity(self, model_dir):
sentence1 = '今天气温比昨天高么?'
sentence2 = '今天湿度比昨天高么?'
model = Model.from_pretrained(model_dir)
pipeline_ins = pipeline(task=Tasks.sentence_similarity, model=model)
print(pipeline_ins(input=(sentence1, sentence2)))
@unittest.skip
def test_finetune_afqmc(self):
"""This unittest is used to reproduce the clue:afqmc dataset + plug meantal model training results.
User can train a custom dataset by modifying this piece of code and comment the @unittest.skip.
"""
def cfg_modify_fn(cfg):
cfg.task = Tasks.sentence_similarity
cfg['preprocessor'] = {'type': Preprocessors.sen_sim_tokenizer}
cfg.train.optimizer.lr = 2e-5
cfg['dataset'] = {
'train': {
'labels': ['0', '1'],
'first_sequence': 'sentence1',
'second_sequence': 'sentence2',
'label': 'label',
}
}
cfg.train.lr_scheduler.total_iters = int(
len(dataset['train']) / 32) * cfg.train.max_epochs
return cfg
dataset = MsDataset.load('clue', subset_name='afqmc')
self.finetune(
model_id='damo/nlp_plug-mental_backbone_base',
train_dataset=dataset['train'],
eval_dataset=dataset['validation'],
cfg_modify_fn=cfg_modify_fn)
output_dir = os.path.join(self.tmp_dir, ModelFile.TRAIN_OUTPUT_DIR)
self.pipeline_sentence_similarity(output_dir)
if __name__ == '__main__':
unittest.main()