From c1651b38935292569b13c3f702a54b65605ce5f5 Mon Sep 17 00:00:00 2001 From: Hongwei Chen Date: Sat, 26 Sep 2026 06:08:31 +0000 Subject: [PATCH 1/3] Remove examples for deprecated features Signed-off-by: Hongwei Chen --- .../dschat/utils/module/lora.py | 2 +- .../DeepSpeed-VisualChat/utils/module/lora.py | 2 +- training/BingBertGlue/nvidia/modelingpreln.py | 78 +- .../nvidia/modelingpreln_layerdrop.py | 1662 ----------------- .../run_glue_classifier_bert_base.py | 13 +- .../run_glue_classifier_bert_large.py | 14 +- training/BingBertGlue/turing/models.py | 7 +- .../deepspeed_onebitadam_bsz96_config.json | 20 - .../run_squad_deepspeed_onebitadam.sh | 60 - .../mpi_ethernet/run_squad_mpi_onebitadam.sh | 60 - .../deepspeed_onebitadam_bsz96_config.json | 20 - .../run_squad_deepspeed_onebitadam.sh | 59 - .../run_squad_mpi_onebitadam.sh | 59 - .../deepspeed_onebitadam_bsz96_config.json | 20 - .../nccl/run_squad_deepspeed_onebitadam.sh | 56 - .../turing/modelingpreln_layerdrop.py | 1652 ---------------- training/MoQ/README.md | 5 - .../research_projects/lxmert/requirements.txt | 98 - training/MoQ/requirements.txt | 3 - training/MoQ/run.sh | 42 - training/MoQ/run_glue.py | 561 ------ training/MoQ/test.json | 25 - ...z4k_01adam_config_seq128_mpi_ethernet.json | 27 - ...z4k_01adam_config_seq512_mpi_ethernet.json | 27 - ...n_bert_01adam_bsz4k_seq128_mpi_ethernet.sh | 31 - ...n_bert_01adam_bsz4k_seq512_mpi_ethernet.sh | 37 - ...k_01adam_config_seq128_mpi_infiniband.json | 27 - ...k_01adam_config_seq512_mpi_infiniband.json | 27 - ...bert_01adam_bsz4k_seq128_mpi_infiniband.sh | 30 - ...bert_01adam_bsz4k_seq512_mpi_infiniband.sh | 36 - ...speed_bsz4k_01adam_config_seq128_nccl.json | 27 - ...speed_bsz4k_01adam_config_seq512_nccl.json | 27 - .../ds_train_bert_01adam_bsz4k_seq128_nccl.sh | 32 - .../ds_train_bert_01adam_bsz4k_seq512_nccl.sh | 38 - ...onebitadam_config_seq128_mpi_ethernet.json | 26 - ...rt_onebitadam_bsz4k_seq128_mpi_ethernet.sh | 33 - ...n_bert_onebitadam_bsz4k_seq128_ethernet.sh | 39 - ...ebitadam_config_seq128_mpi_infiniband.json | 26 - ..._onebitadam_bsz4k_seq128_mpi_infiniband.sh | 32 - ...bert_onebitadam_bsz4k_seq128_infiniband.sh | 38 - ...d_bsz4k_onebitadam_config_seq128_nccl.json | 26 - ...train_bert_onebitadam_bsz4k_seq128_nccl.sh | 29 - ...onebitlamb_config_seq512_mpi_ethernet.json | 31 - ...onebitlamb_config_seq128_mpi_ethernet.json | 32 - ...t_onebitlamb_bsz32k_seq512_mpi_ethernet.sh | 45 - ...t_onebitlamb_bsz64k_seq128_mpi_ethernet.sh | 33 - ..._bert_onebitlamb_bsz32k_seq512_ethernet.sh | 51 - ..._bert_onebitlamb_bsz64k_seq128_ethernet.sh | 39 - ...ebitlamb_config_seq512_mpi_infiniband.json | 31 - ...ebitlamb_config_seq128_mpi_infiniband.json | 32 - ...onebitlamb_bsz32k_seq512_mpi_infiniband.sh | 44 - ...onebitlamb_bsz64k_seq128_mpi_infiniband.sh | 32 - ...ert_onebitlamb_bsz32k_seq512_infiniband.sh | 50 - ...ert_onebitlamb_bsz64k_seq128_infiniband.sh | 39 - ..._bsz32k_onebitlamb_config_seq512_nccl.json | 31 - ..._bsz64k_onebitlamb_config_seq128_nccl.json | 32 - ...rain_bert_onebitlamb_bsz32k_seq512_nccl.sh | 42 - ...rain_bert_onebitlamb_bsz64k_seq128_nccl.sh | 30 - ..._progressive_layer_drop_config_seq128.json | 26 - .../deepspeed_bsz64k_lamb_config_seq128.json | 10 - training/bing_bert/deepspeed_train.py | 83 +- .../ds_sa_train_bert_bsz64k_seq128.sh | 25 - ...ert_progressive_layer_drop_bsz4k_seq128.sh | 25 - training/bing_bert/nvidia/modelingpreln.py | 78 +- .../nvidia/modelingpreln_layerdrop.py | 1662 ----------------- .../bing_bert/run_glue_bert_base_finetune.sh | 2 +- .../run_glue_classifier_bert_base.py | 13 +- .../run_glue_classifier_bert_large.py | 14 +- training/bing_bert/turing/models.py | 7 +- training/bing_bert/utils.py | 10 - 70 files changed, 26 insertions(+), 7656 deletions(-) delete mode 100755 training/BingBertGlue/nvidia/modelingpreln_layerdrop.py delete mode 100644 training/BingBertSquad/1-bit_adam/mpi_ethernet/deepspeed_onebitadam_bsz96_config.json delete mode 100644 training/BingBertSquad/1-bit_adam/mpi_ethernet/run_squad_deepspeed_onebitadam.sh delete mode 100644 training/BingBertSquad/1-bit_adam/mpi_ethernet/run_squad_mpi_onebitadam.sh delete mode 100755 training/BingBertSquad/1-bit_adam/mpi_infiniband/deepspeed_onebitadam_bsz96_config.json delete mode 100755 training/BingBertSquad/1-bit_adam/mpi_infiniband/run_squad_deepspeed_onebitadam.sh delete mode 100755 training/BingBertSquad/1-bit_adam/mpi_infiniband/run_squad_mpi_onebitadam.sh delete mode 100644 training/BingBertSquad/1-bit_adam/nccl/deepspeed_onebitadam_bsz96_config.json delete mode 100644 training/BingBertSquad/1-bit_adam/nccl/run_squad_deepspeed_onebitadam.sh delete mode 100755 training/BingBertSquad/turing/modelingpreln_layerdrop.py delete mode 100644 training/MoQ/README.md delete mode 100644 training/MoQ/huggingface-transformers/examples/research_projects/lxmert/requirements.txt delete mode 100644 training/MoQ/requirements.txt delete mode 100644 training/MoQ/run.sh delete mode 100644 training/MoQ/run_glue.py delete mode 100644 training/MoQ/test.json delete mode 100644 training/bing_bert/01_adam/mpi_ethernet/deepspeed_bsz4k_01adam_config_seq128_mpi_ethernet.json delete mode 100644 training/bing_bert/01_adam/mpi_ethernet/deepspeed_bsz4k_01adam_config_seq512_mpi_ethernet.json delete mode 100644 training/bing_bert/01_adam/mpi_ethernet/ds_train_bert_01adam_bsz4k_seq128_mpi_ethernet.sh delete mode 100644 training/bing_bert/01_adam/mpi_ethernet/ds_train_bert_01adam_bsz4k_seq512_mpi_ethernet.sh delete mode 100644 training/bing_bert/01_adam/mpi_infiniband/deepspeed_bsz4k_01adam_config_seq128_mpi_infiniband.json delete mode 100644 training/bing_bert/01_adam/mpi_infiniband/deepspeed_bsz4k_01adam_config_seq512_mpi_infiniband.json delete mode 100644 training/bing_bert/01_adam/mpi_infiniband/ds_train_bert_01adam_bsz4k_seq128_mpi_infiniband.sh delete mode 100644 training/bing_bert/01_adam/mpi_infiniband/ds_train_bert_01adam_bsz4k_seq512_mpi_infiniband.sh delete mode 100644 training/bing_bert/01_adam/nccl/deepspeed_bsz4k_01adam_config_seq128_nccl.json delete mode 100644 training/bing_bert/01_adam/nccl/deepspeed_bsz4k_01adam_config_seq512_nccl.json delete mode 100644 training/bing_bert/01_adam/nccl/ds_train_bert_01adam_bsz4k_seq128_nccl.sh delete mode 100644 training/bing_bert/01_adam/nccl/ds_train_bert_01adam_bsz4k_seq512_nccl.sh delete mode 100644 training/bing_bert/1-bit_adam/mpi_ethernet/deepspeed_bsz4k_onebitadam_config_seq128_mpi_ethernet.json delete mode 100644 training/bing_bert/1-bit_adam/mpi_ethernet/ds_train_bert_onebitadam_bsz4k_seq128_mpi_ethernet.sh delete mode 100644 training/bing_bert/1-bit_adam/mpi_ethernet/mpi_train_bert_onebitadam_bsz4k_seq128_ethernet.sh delete mode 100644 training/bing_bert/1-bit_adam/mpi_infiniband/deepspeed_bsz4k_onebitadam_config_seq128_mpi_infiniband.json delete mode 100644 training/bing_bert/1-bit_adam/mpi_infiniband/ds_train_bert_onebitadam_bsz4k_seq128_mpi_infiniband.sh delete mode 100644 training/bing_bert/1-bit_adam/mpi_infiniband/mpi_train_bert_onebitadam_bsz4k_seq128_infiniband.sh delete mode 100644 training/bing_bert/1-bit_adam/nccl/deepspeed_bsz4k_onebitadam_config_seq128_nccl.json delete mode 100644 training/bing_bert/1-bit_adam/nccl/ds_train_bert_onebitadam_bsz4k_seq128_nccl.sh delete mode 100644 training/bing_bert/1-bit_lamb/mpi_ethernet/deepspeed_bsz32k_onebitlamb_config_seq512_mpi_ethernet.json delete mode 100644 training/bing_bert/1-bit_lamb/mpi_ethernet/deepspeed_bsz64k_onebitlamb_config_seq128_mpi_ethernet.json delete mode 100644 training/bing_bert/1-bit_lamb/mpi_ethernet/ds_train_bert_onebitlamb_bsz32k_seq512_mpi_ethernet.sh delete mode 100644 training/bing_bert/1-bit_lamb/mpi_ethernet/ds_train_bert_onebitlamb_bsz64k_seq128_mpi_ethernet.sh delete mode 100644 training/bing_bert/1-bit_lamb/mpi_ethernet/mpi_train_bert_onebitlamb_bsz32k_seq512_ethernet.sh delete mode 100644 training/bing_bert/1-bit_lamb/mpi_ethernet/mpi_train_bert_onebitlamb_bsz64k_seq128_ethernet.sh delete mode 100644 training/bing_bert/1-bit_lamb/mpi_infiniband/deepspeed_bsz32k_onebitlamb_config_seq512_mpi_infiniband.json delete mode 100644 training/bing_bert/1-bit_lamb/mpi_infiniband/deepspeed_bsz64k_onebitlamb_config_seq128_mpi_infiniband.json delete mode 100644 training/bing_bert/1-bit_lamb/mpi_infiniband/ds_train_bert_onebitlamb_bsz32k_seq512_mpi_infiniband.sh delete mode 100644 training/bing_bert/1-bit_lamb/mpi_infiniband/ds_train_bert_onebitlamb_bsz64k_seq128_mpi_infiniband.sh delete mode 100644 training/bing_bert/1-bit_lamb/mpi_infiniband/mpi_train_bert_onebitlamb_bsz32k_seq512_infiniband.sh delete mode 100644 training/bing_bert/1-bit_lamb/mpi_infiniband/mpi_train_bert_onebitlamb_bsz64k_seq128_infiniband.sh delete mode 100644 training/bing_bert/1-bit_lamb/nccl/deepspeed_bsz32k_onebitlamb_config_seq512_nccl.json delete mode 100644 training/bing_bert/1-bit_lamb/nccl/deepspeed_bsz64k_onebitlamb_config_seq128_nccl.json delete mode 100644 training/bing_bert/1-bit_lamb/nccl/ds_train_bert_onebitlamb_bsz32k_seq512_nccl.sh delete mode 100644 training/bing_bert/1-bit_lamb/nccl/ds_train_bert_onebitlamb_bsz64k_seq128_nccl.sh delete mode 100755 training/bing_bert/deepspeed_bsz4k_progressive_layer_drop_config_seq128.json delete mode 100644 training/bing_bert/ds_sa_train_bert_bsz64k_seq128.sh delete mode 100755 training/bing_bert/ds_train_bert_progressive_layer_drop_bsz4k_seq128.sh delete mode 100755 training/bing_bert/nvidia/modelingpreln_layerdrop.py diff --git a/applications/DeepSpeed-Chat/dschat/utils/module/lora.py b/applications/DeepSpeed-Chat/dschat/utils/module/lora.py index 32c9730b6..056ab9b33 100644 --- a/applications/DeepSpeed-Chat/dschat/utils/module/lora.py +++ b/applications/DeepSpeed-Chat/dschat/utils/module/lora.py @@ -6,7 +6,7 @@ import torch from torch import nn import torch.nn.functional as F -from deepspeed.compression.helper import recursive_getattr, recursive_setattr +from deepspeed.utils.module_utils import recursive_getattr, recursive_setattr import deepspeed diff --git a/applications/DeepSpeed-VisualChat/utils/module/lora.py b/applications/DeepSpeed-VisualChat/utils/module/lora.py index 67e446033..c2767f340 100644 --- a/applications/DeepSpeed-VisualChat/utils/module/lora.py +++ b/applications/DeepSpeed-VisualChat/utils/module/lora.py @@ -6,7 +6,7 @@ import torch from torch import nn import torch.nn.functional as F -from deepspeed.compression.helper import recursive_getattr, recursive_setattr +from deepspeed.utils.module_utils import recursive_getattr, recursive_setattr import deepspeed diff --git a/training/BingBertGlue/nvidia/modelingpreln.py b/training/BingBertGlue/nvidia/modelingpreln.py index 69c650b56..db5ea1d1a 100755 --- a/training/BingBertGlue/nvidia/modelingpreln.py +++ b/training/BingBertGlue/nvidia/modelingpreln.py @@ -75,44 +75,6 @@ def get_deepspeed_config(args): raise RuntimeError('deepspeed_config is not found in args.') -def get_sparse_attention_config(args, num_heads): - if args.deepspeed_sparse_attention: - ds_config = get_deepspeed_config(args) - if hasattr(ds_config, - 'sparse_attention') and ds_config.sparse_attention: - sa_config = ds_config.sparse_attention - sa_mode = sa_config.get('mode') - if (sa_mode == 'dense'): - from deepspeed.ops.sparse_attention import DenseSparsityConfig as STConfig - elif (sa_mode == 'fixed'): - from deepspeed.ops.sparse_attention import FixedSparsityConfig as STConfig - elif (sa_mode == 'bigbird'): - from deepspeed.ops.sparse_attention import BigBirdSparsityConfig as STConfig - elif (sa_mode == 'bslongformer'): - from deepspeed.ops.sparse_attention import BSLongformerSparsityConfig as STConfig - elif (sa_mode == 'variable'): - from deepspeed.ops.sparse_attention import VariableSparsityConfig as STConfig - else: - raise NotImplementedError( - f'Given sparsity mode, {sa_mode}, has not been implemented yet!' - ) - del sa_config['mode'] - return STConfig(num_heads=num_heads, **sa_config) - else: - from deepspeed.ops.sparse_attention import FixedSparsityConfig as STConfig - print( - 'deepspeed sparse attention is not set; Fixed sparsity is used as default.' - ) - return STConfig(num_heads=num_heads) - else: - return None - -def get_sparse_attention_utils(sparse_attention_config): - if sparse_attention_config is not None: - from deepspeed.ops.sparse_attention import SparseAttentionUtils - return SparseAttentionUtils - return None - def load_tf_weights_in_bert(model, tf_checkpoint_path): """ Load tf checkpoints in a pytorch model """ @@ -557,17 +519,12 @@ def forward(self, hidden_states, attention_mask): class BertEncoder(nn.Module): - def __init__(self, config, args, sparse_attention_config=None): + def __init__(self, config, args): super(BertEncoder, self).__init__() #Added later to make it similar to GPT-2 self.FinalLayerNorm = BertLayerNorm(config.hidden_size, eps=1e-12) - if args.deepspeed_transformer_kernel and args.deepspeed_sparse_attention: - raise NotImplementedError( - f'Currently DeepSpeed Transformer Kernels do not support Sparse Attention. To use Sparse Attention, you need to disable Transformer Kernels!' - ) - if args.deepspeed_transformer_kernel: from deepspeed import DeepSpeedTransformerLayer, DeepSpeedTransformerConfig @@ -601,12 +558,6 @@ def __init__(self, config, args, sparse_attention_config=None): ]) else: layer = BertLayer(config) - if sparse_attention_config is not None: - from deepspeed.ops.sparse_attention import BertSparseSelfAttention - - layer.attention.self = BertSparseSelfAttention( - config, sparsity_config=sparse_attention_config) - self.layer = nn.ModuleList([ copy.deepcopy(layer) for _ in range(config.num_hidden_layers) ]) @@ -995,15 +946,7 @@ class BertModel(BertPreTrainedModel): def __init__(self, config, args=None): super(BertModel, self).__init__(config) self.embeddings = BertEmbeddings(config) - # set pad_token_id that is used for sparse attention padding - self.pad_token_id = config.pad_token_id if hasattr( - config, 'pad_token_id') and config.pad_token_id is not None else 0 - # set sparse_attention_config if it has been selected - self.sparse_attention_config = get_sparse_attention_config( - args, config.num_attention_heads) - self.sparse_attention_utils = get_sparse_attention_utils(self.sparse_attention_config) - self.encoder = BertEncoder( - config, args, sparse_attention_config=self.sparse_attention_config) + self.encoder = BertEncoder(config, args) self.pooler = BertPooler(config) self.apply(self.init_bert_weights) logger.info("Init BERT pretrain model") @@ -1035,18 +978,6 @@ def forward(self, dtype=next(self.parameters()).dtype) # fp16 compatibility extended_attention_mask = (1.0 - extended_attention_mask) * -10000.0 - # If BertEncoder uses sparse attention, it needs to be padded based on the sparse attention block size - if self.sparse_attention_config is not None: - pad_len, input_ids, attention_mask, token_type_ids, position_ids, inputs_embeds = self.sparse_attention_utils.pad_to_block_size( - block_size=self.sparse_attention_config.block, - input_ids=input_ids, - attention_mask=extended_attention_mask, - token_type_ids=token_type_ids, - position_ids=None, - inputs_embeds=None, - pad_token_id=self.pad_token_id, - model_mbeddings=self.embeddings) - embedding_output = self.embeddings(input_ids, token_type_ids) encoded_layers = self.encoder( embedding_output, @@ -1056,11 +987,6 @@ def forward(self, sequence_output = encoded_layers[-1] pooled_output = self.pooler(sequence_output) - # If BertEncoder uses sparse attention, and input_ids were padded, sequence output needs to be unpadded to original length - if self.sparse_attention_config is not None and pad_len > 0: - encoded_layers[-1] = self.sparse_attention_utils.unpad_sequence_output( - pad_len, encoded_layers[-1]) - if not output_all_encoded_layers: encoded_layers = encoded_layers[-1] return encoded_layers, pooled_output diff --git a/training/BingBertGlue/nvidia/modelingpreln_layerdrop.py b/training/BingBertGlue/nvidia/modelingpreln_layerdrop.py deleted file mode 100755 index b5beb89af..000000000 --- a/training/BingBertGlue/nvidia/modelingpreln_layerdrop.py +++ /dev/null @@ -1,1662 +0,0 @@ -# DeepSpeed note, code taken from commit 3d59216cec89a363649b4fe3d15295ba936ced0f -# https://github.com/NVIDIA/DeepLearningExamples/blob/master/PyTorch/LanguageModeling/BERT/modeling.py - -# coding=utf-8 -# Copyright 2018 The Google AI Language Team Authors and The HugginFace Inc. team. -# Copyright (c) 2018, NVIDIA CORPORATION. All rights reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -"""PyTorch BERT model.""" - -from __future__ import absolute_import, division, print_function, unicode_literals - -import copy -import json -import logging -import math -import os -import shutil -import tarfile -import tempfile -import sys -from io import open - -import torch -from torch import nn -from torch.nn import CrossEntropyLoss -from torch.utils import checkpoint - -from turing.file_utils import cached_path - -from torch.nn import Module -from torch.nn.parameter import Parameter -import torch.nn.functional as F -import torch.nn.init as init - -import numpy as np - -logger = logging.getLogger(__name__) - -PRETRAINED_MODEL_ARCHIVE_MAP = { - 'bert-base-uncased': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-base-uncased.tar.gz", - 'bert-large-uncased': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-large-uncased.tar.gz", - 'bert-base-cased': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-base-cased.tar.gz", - 'bert-large-cased': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-large-cased.tar.gz", - 'bert-base-multilingual-uncased': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-base-multilingual-uncased.tar.gz", - 'bert-base-multilingual-cased': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-base-multilingual-cased.tar.gz", - 'bert-base-chinese': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-base-chinese.tar.gz", -} -CONFIG_NAME = 'bert_config.json' -WEIGHTS_NAME = 'pytorch_model.bin' -TF_WEIGHTS_NAME = 'model.ckpt' - - -def get_deepspeed_config(args): - if hasattr(args, 'deepspeed_config') and args.deepspeed_config: - from deepspeed import DeepSpeedConfig - return DeepSpeedConfig(args.deepspeed_config) - else: - raise RuntimeError('deepspeed_config is not found in args.') - - -def get_sparse_attention_config(args, num_heads): - if args.deepspeed_sparse_attention: - ds_config = get_deepspeed_config(args) - if hasattr(ds_config, - 'sparse_attention') and ds_config.sparse_attention: - sa_config = ds_config.sparse_attention - sa_mode = sa_config.get('mode') - if (sa_mode == 'dense'): - from deepspeed.ops.sparse_attention import DenseSparsityConfig as STConfig - elif (sa_mode == 'fixed'): - from deepspeed.ops.sparse_attention import FixedSparsityConfig as STConfig - elif (sa_mode == 'bigbird'): - from deepspeed.ops.sparse_attention import BigBirdSparsityConfig as STConfig - elif (sa_mode == 'bslongformer'): - from deepspeed.ops.sparse_attention import BSLongformerSparsityConfig as STConfig - elif (sa_mode == 'variable'): - from deepspeed.ops.sparse_attention import VariableSparsityConfig as STConfig - else: - raise NotImplementedError( - f'Given sparsity mode, {sa_mode}, has not been implemented yet!' - ) - del sa_config['mode'] - return STConfig(num_heads=num_heads, **sa_config) - else: - from deepspeed.ops.sparse_attention import FixedSparsityConfig as STConfig - print( - 'deepspeed sparse attention is not set; Fixed sparsity is used as default.' - ) - return STConfig(num_heads=num_heads) - else: - return None - - -def get_sparse_attention_utils(sparse_attention_config): - if sparse_attention_config is not None: - from deepspeed.ops.sparse_attention import SparseAttentionUtils - return SparseAttentionUtils - return None - - -def load_tf_weights_in_bert(model, tf_checkpoint_path): - """ Load tf checkpoints in a pytorch model - """ - try: - import re - import numpy as np - import tensorflow as tf - except ImportError: - print( - "Loading a TensorFlow models in PyTorch, requires TensorFlow to be installed. Please see " - "https://www.tensorflow.org/install/ for installation instructions." - ) - raise - tf_path = os.path.abspath(tf_checkpoint_path) - print("Converting TensorFlow checkpoint from {}".format(tf_path)) - # Load weights from TF model - init_vars = tf.train.list_variables(tf_path) - names = [] - arrays = [] - for name, shape in init_vars: - print("Loading TF weight {} with shape {}".format(name, shape)) - array = tf.train.load_variable(tf_path, name) - names.append(name) - arrays.append(array) - - for name, array in zip(names, arrays): - name = name.split('/') - # adam_v and adam_m are variables used in AdamWeightDecayOptimizer to calculated m and v - # which are not required for using pretrained model - if any(n in ["adam_v", "adam_m"] for n in name): - print("Skipping {}".format("/".join(name))) - continue - pointer = model - for m_name in name: - if re.fullmatch(r'[A-Za-z]+_\d+', m_name): - l = re.split(r'_(\d+)', m_name) - else: - l = [m_name] - if l[0] == 'kernel' or l[0] == 'gamma': - pointer = getattr(pointer, 'weight') - elif l[0] == 'output_bias' or l[0] == 'beta': - pointer = getattr(pointer, 'bias') - elif l[0] == 'output_weights': - pointer = getattr(pointer, 'weight') - else: - pointer = getattr(pointer, l[0]) - if len(l) >= 2: - num = int(l[1]) - pointer = pointer[num] - if m_name[-11:] == '_embeddings': - pointer = getattr(pointer, 'weight') - elif m_name == 'kernel': - array = np.transpose(array) - try: - assert pointer.shape == array.shape - except AssertionError as e: - e.args += (pointer.shape, array.shape) - raise - print("Initialize PyTorch weight {}".format(name)) - pointer.data = torch.from_numpy(array) - return model - - -@torch.jit.script -def f_gelu(x): - pdtype = x.dtype - x = x.float() - y = x * 0.5 * (1.0 + torch.erf(x / math.sqrt(2.0))) - return y.to(pdtype) - - -@torch.jit.script -def bias_gelu(bias, y): - x = bias + y - return x * 0.5 * (1.0 + torch.erf(x / 1.41421)) - - -@torch.jit.script -def bias_tanh(bias, y): - x = bias + y - return torch.tanh(x) - - -def gelu(x): - """Implementation of the gelu activation function. - For information: OpenAI GPT's gelu is slightly different (and gives slightly different results): - 0.5 * x * (1 + torch.tanh(math.sqrt(2 / math.pi) * (x + 0.044715 * torch.pow(x, 3)))) - Also see https://arxiv.org/abs/1606.08415 - """ - return f_gelu(x) - - -def swish(x): - return x * torch.sigmoid(x) - - -ACT2FN = {"gelu": gelu, "relu": torch.nn.functional.relu, "swish": swish} - - -class LinearActivation(Module): - r"""Fused Linear and activation Module. - """ - __constants__ = ['bias'] - - def __init__(self, in_features, out_features, act='gelu', bias=True): - super(LinearActivation, self).__init__() - self.in_features = in_features - self.out_features = out_features - self.fused_gelu = False - self.fused_tanh = False - if isinstance(act, str) or (sys.version_info[0] == 2 - and isinstance(act, unicode)): - if bias and act == 'gelu': - self.fused_gelu = True - elif bias and act == 'tanh': - self.fused_tanh = True - else: - self.act_fn = ACT2FN[act] - else: - self.act_fn = act - self.weight = Parameter(torch.Tensor(out_features, in_features)) - if bias: - self.bias = Parameter(torch.Tensor(out_features)) - else: - self.register_parameter('bias', None) - self.reset_parameters() - - def reset_parameters(self): - init.kaiming_uniform_(self.weight, a=math.sqrt(5)) - if self.bias is not None: - fan_in, _ = init._calculate_fan_in_and_fan_out(self.weight) - bound = 1 / math.sqrt(fan_in) - init.uniform_(self.bias, -bound, bound) - - def forward(self, input): - if self.fused_gelu: - return bias_gelu(self.bias, F.linear(input, self.weight, None)) - elif self.fused_tanh: - return bias_tanh(self.bias, F.linear(input, self.weight, None)) - else: - return self.act_fn(F.linear(input, self.weight, self.bias)) - - def extra_repr(self): - return 'in_features={}, out_features={}, bias={}'.format( - self.in_features, self.out_features, self.bias is not None) - - -class BertConfig(object): - """Configuration class to store the configuration of a `BertModel`. - """ - def __init__(self, - vocab_size_or_config_json_file, - hidden_size=768, - num_hidden_layers=12, - num_attention_heads=12, - intermediate_size=3072, - hidden_act="gelu", - hidden_dropout_prob=0.1, - attention_probs_dropout_prob=0.1, - max_position_embeddings=512, - type_vocab_size=2, - initializer_range=0.02): - """Constructs BertConfig. - - Args: - vocab_size_or_config_json_file: Vocabulary size of `inputs_ids` in `BertModel`. - hidden_size: Size of the encoder layers and the pooler layer. - num_hidden_layers: Number of hidden layers in the Transformer encoder. - num_attention_heads: Number of attention heads for each attention layer in - the Transformer encoder. - intermediate_size: The size of the "intermediate" (i.e., feed-forward) - layer in the Transformer encoder. - hidden_act: The non-linear activation function (function or string) in the - encoder and pooler. If string, "gelu", "relu" and "swish" are supported. - hidden_dropout_prob: The dropout probabilitiy for all fully connected - layers in the embeddings, encoder, and pooler. - attention_probs_dropout_prob: The dropout ratio for the attention - probabilities. - max_position_embeddings: The maximum sequence length that this model might - ever be used with. Typically set this to something large just in case - (e.g., 512 or 1024 or 2048). - type_vocab_size: The vocabulary size of the `token_type_ids` passed into - `BertModel`. - initializer_range: The sttdev of the truncated_normal_initializer for - initializing all weight matrices. - """ - if isinstance(vocab_size_or_config_json_file, - str) or (sys.version_info[0] == 2 and isinstance( - vocab_size_or_config_json_file, unicode)): - with open(vocab_size_or_config_json_file, "r", - encoding='utf-8') as reader: - json_config = json.loads(reader.read()) - for key, value in json_config.items(): - self.__dict__[key] = value - elif isinstance(vocab_size_or_config_json_file, int): - self.vocab_size = vocab_size_or_config_json_file - self.hidden_size = hidden_size - self.num_hidden_layers = num_hidden_layers - self.num_attention_heads = num_attention_heads - self.hidden_act = hidden_act - self.intermediate_size = intermediate_size - self.hidden_dropout_prob = hidden_dropout_prob - self.attention_probs_dropout_prob = attention_probs_dropout_prob - self.max_position_embeddings = max_position_embeddings - self.type_vocab_size = type_vocab_size - self.initializer_range = initializer_range - else: - raise ValueError( - "First argument must be either a vocabulary size (int)" - "or the path to a pretrained model config file (str)") - - @classmethod - def from_dict(cls, json_object): - """Constructs a `BertConfig` from a Python dictionary of parameters.""" - config = BertConfig(vocab_size_or_config_json_file=-1) - for key, value in json_object.items(): - config.__dict__[key] = value - return config - - @classmethod - def from_json_file(cls, json_file): - """Constructs a `BertConfig` from a json file of parameters.""" - with open(json_file, "r", encoding='utf-8') as reader: - text = reader.read() - return cls.from_dict(json.loads(text)) - - def __repr__(self): - return str(self.to_json_string()) - - def to_dict(self): - """Serializes this instance to a Python dictionary.""" - output = copy.deepcopy(self.__dict__) - return output - - def to_json_string(self): - """Serializes this instance to a JSON string.""" - return json.dumps(self.to_dict(), indent=2, sort_keys=True) + "\n" - - -try: - import apex - #apex.amp.register_half_function(apex.normalization.fused_layer_norm, 'FusedLayerNorm') - import apex.normalization - #apex.amp.register_float_function(apex.normalization.FusedLayerNorm, 'forward') - BertLayerNorm = apex.normalization.FusedLayerNorm -except ImportError: - print( - "Better speed can be achieved with apex installed from https://www.github.com/nvidia/apex." - ) - - class BertLayerNorm(nn.Module): - def __init__(self, hidden_size, eps=1e-12): - """Construct a layernorm module in the TF style (epsilon inside the square root). - """ - super(BertLayerNorm, self).__init__() - self.weight = nn.Parameter(torch.ones(hidden_size)) - self.bias = nn.Parameter(torch.zeros(hidden_size)) - self.variance_epsilon = eps - - def forward(self, x): - pdtype = x.dtype - x = x.float() - u = x.mean(-1, keepdim=True) - s = (x - u).pow(2).mean(-1, keepdim=True) - x = (x - u) / torch.sqrt(s + self.variance_epsilon) - return self.weight * x.to(pdtype) + self.bias - - -class BertEmbeddings(nn.Module): - """Construct the embeddings from word, position and token_type embeddings. - """ - def __init__(self, config): - super(BertEmbeddings, self).__init__() - self.word_embeddings = nn.Embedding(config.vocab_size, - config.hidden_size) - self.position_embeddings = nn.Embedding(config.max_position_embeddings, - config.hidden_size) - self.token_type_embeddings = nn.Embedding(config.type_vocab_size, - config.hidden_size) - - # self.LayerNorm is not snake-cased to stick with TensorFlow model variable name and be able to load - # any TensorFlow checkpoint file - self.LayerNorm = BertLayerNorm(config.hidden_size, eps=1e-12) - self.dropout = nn.Dropout(config.hidden_dropout_prob) - - def forward(self, input_ids, token_type_ids=None): - seq_length = input_ids.size(1) - position_ids = torch.arange(seq_length, - dtype=torch.long, - device=input_ids.device) - position_ids = position_ids.unsqueeze(0).expand_as(input_ids) - if token_type_ids is None: - token_type_ids = torch.zeros_like(input_ids) - - words_embeddings = self.word_embeddings(input_ids) - position_embeddings = self.position_embeddings(position_ids) - token_type_embeddings = self.token_type_embeddings(token_type_ids) - - embeddings = words_embeddings + position_embeddings + token_type_embeddings - embeddings = self.LayerNorm(embeddings) - embeddings = self.dropout(embeddings) - return embeddings - - -class BertSelfAttention(nn.Module): - def __init__(self, config): - super(BertSelfAttention, self).__init__() - if config.hidden_size % config.num_attention_heads != 0: - raise ValueError( - "The hidden size (%d) is not a multiple of the number of attention " - "heads (%d)" % - (config.hidden_size, config.num_attention_heads)) - self.num_attention_heads = config.num_attention_heads - self.attention_head_size = int(config.hidden_size / - config.num_attention_heads) - self.all_head_size = self.num_attention_heads * self.attention_head_size - - self.query = nn.Linear(config.hidden_size, self.all_head_size) - self.key = nn.Linear(config.hidden_size, self.all_head_size) - self.value = nn.Linear(config.hidden_size, self.all_head_size) - - self.dropout = nn.Dropout(config.attention_probs_dropout_prob) - self.softmax = nn.Softmax(dim=-1) - - def transpose_for_scores(self, x): - new_x_shape = x.size()[:-1] + (self.num_attention_heads, - self.attention_head_size) - x = x.view(*new_x_shape) - return x.permute(0, 2, 1, 3) - - def transpose_key_for_scores(self, x): - new_x_shape = x.size()[:-1] + (self.num_attention_heads, - self.attention_head_size) - x = x.view(*new_x_shape) - return x.permute(0, 2, 3, 1) - - def forward(self, hidden_states, attention_mask): - mixed_query_layer = self.query(hidden_states) - mixed_key_layer = self.key(hidden_states) - mixed_value_layer = self.value(hidden_states) - - query_layer = self.transpose_for_scores(mixed_query_layer) - key_layer = self.transpose_key_for_scores(mixed_key_layer) - value_layer = self.transpose_for_scores(mixed_value_layer) - - # Take the dot product between "query" and "key" to get the raw attention scores. - attention_scores = torch.matmul(query_layer, key_layer) - attention_scores = attention_scores / math.sqrt( - self.attention_head_size) - # Apply the attention mask is (precomputed for all layers in BertModel forward() function) - attention_scores = attention_scores + attention_mask - - pdtype = attention_scores.dtype - # Normalize the attention scores to probabilities. - attention_probs = self.softmax(attention_scores) - - # This is actually dropping out entire tokens to attend to, which might - # seem a bit unusual, but is taken from the original Transformer paper. - attention_probs = self.dropout(attention_probs) - - context_layer = torch.matmul(attention_probs, value_layer) - context_layer = context_layer.permute(0, 2, 1, 3).contiguous() - new_context_layer_shape = context_layer.size()[:-2] + ( - self.all_head_size, ) - context_layer = context_layer.view(*new_context_layer_shape) - return context_layer - - -class BertSelfOutput(nn.Module): - def __init__(self, config): - super(BertSelfOutput, self).__init__() - self.dense = nn.Linear(config.hidden_size, config.hidden_size) - self.dense.bert_output_layer = True - self.dropout = nn.Dropout(config.hidden_dropout_prob) - - def forward(self, hidden_states, input_tensor): - hidden_states = self.dense(hidden_states) - hidden_states = self.dropout(hidden_states) - return hidden_states - - -class BertAttention(nn.Module): - def __init__(self, config): - super(BertAttention, self).__init__() - self.self = BertSelfAttention(config) - self.output = BertSelfOutput(config) - - def forward(self, input_tensor, attention_mask): - self_output = self.self(input_tensor, attention_mask) - attention_output = self.output(self_output, input_tensor) - return attention_output - - -class BertIntermediate(nn.Module): - def __init__(self, config): - super(BertIntermediate, self).__init__() - self.dense_act = LinearActivation(config.hidden_size, - config.intermediate_size, - act=config.hidden_act) - - def forward(self, hidden_states): - hidden_states = self.dense_act(hidden_states) - return hidden_states - - -class BertOutput(nn.Module): - def __init__(self, config): - super(BertOutput, self).__init__() - self.dense = nn.Linear(config.intermediate_size, config.hidden_size) - self.dense.bert_output_layer = True - self.dropout = nn.Dropout(config.hidden_dropout_prob) - - def forward(self, hidden_states): - hidden_states = self.dense(hidden_states) - hidden_states = self.dropout(hidden_states) - return hidden_states - - -class BertLayer(nn.Module): - def __init__(self, config): - super(BertLayer, self).__init__() - self.attention = BertAttention(config) - self.PreAttentionLayerNorm = BertLayerNorm(config.hidden_size, - eps=1e-12) - self.PostAttentionLayerNorm = BertLayerNorm(config.hidden_size, - eps=1e-12) - self.intermediate = BertIntermediate(config) - self.output = BertOutput(config) - - def forward(self, hidden_states, attention_mask, action=1, keep_prob=1.0): - if action == 0: - intermediate_input = hidden_states - else: - input_layer_norm = self.PreAttentionLayerNorm(hidden_states) - attention_output = self.attention(input_layer_norm, attention_mask) - attention_output = attention_output * 1 / keep_prob - intermediate_input = hidden_states + attention_output - - if action == 0: - layer_output = intermediate_input - else: - intermediate_layer_norm = self.PostAttentionLayerNorm( - intermediate_input) - intermediate_output = self.intermediate(intermediate_layer_norm) - layer_output = self.output(intermediate_output) - layer_output = layer_output * 1 / keep_prob - layer_output = layer_output + intermediate_input - - return layer_output - - -class BertEncoder(nn.Module): - def __init__(self, config, args, sparse_attention_config=None): - super(BertEncoder, self).__init__() - - #Added later to make it similar to GPT-2 - self.FinalLayerNorm = BertLayerNorm(config.hidden_size, eps=1e-12) - - if args.deepspeed_transformer_kernel and args.deepspeed_sparse_attention: - raise NotImplementedError( - f'Currently DeepSpeed Transformer Kernels do not support Sparse Attention. To use Sparse Attention, you need to disable Transformer Kernels!' - ) - - if args.deepspeed_transformer_kernel: - from deepspeed import DeepSpeedTransformerLayer, DeepSpeedTransformerConfig - - ds_config = get_deepspeed_config(args) - cuda_config = DeepSpeedTransformerConfig( - batch_size=ds_config.train_micro_batch_size_per_gpu, - max_seq_length=args.max_seq_length, - hidden_size=config.hidden_size, - intermediate_size=config.intermediate_size, - heads=config.num_attention_heads, - attn_dropout_ratio=config.attention_probs_dropout_prob, - hidden_dropout_ratio=config.hidden_dropout_prob, - num_hidden_layers=config.num_hidden_layers, - initializer_range=config.initializer_range, - local_rank=args.local_rank - if hasattr(args, 'local_rank') else -1, - seed=args.seed, - fp16=ds_config.fp16_enabled, - pre_layer_norm=True, - attn_dropout_checkpoint=args.attention_dropout_checkpoint, - normalize_invertible=args.normalize_invertible, - gelu_checkpoint=args.gelu_checkpoint, - stochastic_mode=args.stochastic_mode) - - self.layer = nn.ModuleList([ - copy.deepcopy(DeepSpeedTransformerLayer(i, cuda_config)) - for i in range(config.num_hidden_layers) - ]) - else: - layer = BertLayer(config) - if sparse_attention_config is not None: - from deepspeed.ops.sparse_attention import BertSparseSelfAttention - - layer.attention.self = BertSparseSelfAttention( - config, sparsity_config=sparse_attention_config) - - self.layer = nn.ModuleList([ - copy.deepcopy(layer) for _ in range(config.num_hidden_layers) - ]) - - # def forward(self, hidden_states, attention_mask, output_all_encoded_layers=True): - # all_encoder_layers = [] - # for layer_module in self.layer: - # hidden_states = layer_module(hidden_states, attention_mask) - # if output_all_encoded_layers: - # all_encoder_layers.append(hidden_states) - # if not output_all_encoded_layers: - # all_encoder_layers.append(hidden_states) - # return all_encoder_layers - def forward(self, - hidden_states, - attention_mask, - output_all_encoded_layers=True, - checkpoint_activations=False, - progressive_layer_drop=False, - theta=0.5): - all_encoder_layers = [] - - def custom(start, end): - def custom_forward(*inputs): - layers = self.layer[start:end] - x_ = inputs[0] - for layer in layers: - x_ = layer(x_, inputs[1]) - return x_ - - return custom_forward - - if checkpoint_activations: - l = 0 - num_layers = len(self.layer) - chunk_length = math.ceil(math.sqrt(num_layers)) - while l < num_layers: - hidden_states = checkpoint.checkpoint( - custom(l, l + chunk_length), hidden_states, - attention_mask * 1) - l += chunk_length - # decoder layers - else: - if not progressive_layer_drop: - for i, layer_module in enumerate(self.layer): - hidden_states = layer_module(hidden_states, attention_mask) - - if output_all_encoded_layers: - all_encoder_layers.append(hidden_states) - else: - drop_prob = 1 - theta - step = drop_prob / len(self.layer) - p = 1.0 - # print("+ stochastic drop, depth, Theta {}:".format(theta)) - - for i, layer_module in enumerate(self.layer): - - action = np.random.choice([1, 0], p=[p, 1 - p]) - p = p - step - hidden_states = layer_module(hidden_states, attention_mask, - action, p) - if output_all_encoded_layers: - all_encoder_layers.append(hidden_states) - - if not output_all_encoded_layers or checkpoint_activations: - hidden_states = self.FinalLayerNorm(hidden_states) - all_encoder_layers.append(hidden_states) - return all_encoder_layers - - -#class BertEncoder(nn.Module): -# def __init__(self, config): -# super(BertEncoder, self).__init__() -# layer = BertLayer(config) -# self.layer = nn.ModuleList([copy.deepcopy(layer) for _ in range(config.num_hidden_layers)]) -# -# def forward(self, hidden_states, attention_mask, output_all_encoded_layers=True): -# all_encoder_layers = [] -# for layer_module in self.layer: -# hidden_states = layer_module(hidden_states, attention_mask) -# if output_all_encoded_layers: -# all_encoder_layers.append(hidden_states) -# if not output_all_encoded_layers: -# all_encoder_layers.append(hidden_states) -# return all_encoder_layers - - -class BertPooler(nn.Module): - def __init__(self, config): - super(BertPooler, self).__init__() - self.dense_act = LinearActivation(config.hidden_size, - config.hidden_size, - act="tanh") - - def forward(self, hidden_states): - # We "pool" the model by simply taking the hidden state corresponding - # to the first token. - first_token_tensor = hidden_states[:, 0] - pooled_output = self.dense_act(first_token_tensor) - return pooled_output - - -class BertPredictionHeadTransform(nn.Module): - def __init__(self, config): - super(BertPredictionHeadTransform, self).__init__() - self.dense_act = LinearActivation(config.hidden_size, - config.hidden_size, - act=config.hidden_act) - self.LayerNorm = BertLayerNorm(config.hidden_size, eps=1e-12) - - def forward(self, hidden_states): - hidden_states = self.dense_act(hidden_states) - hidden_states = self.LayerNorm(hidden_states) - return hidden_states - - -class BertLMPredictionHead(nn.Module): - def __init__(self, config, bert_model_embedding_weights): - super(BertLMPredictionHead, self).__init__() - self.transform = BertPredictionHeadTransform(config) - - # The output weights are the same as the input embeddings, but there is - # an output-only bias for each token. - self.decoder = nn.Linear(bert_model_embedding_weights.size(1), - bert_model_embedding_weights.size(0), - bias=False) - self.decoder.weight = bert_model_embedding_weights - self.bias = nn.Parameter( - torch.zeros(bert_model_embedding_weights.size(0))) - - def forward(self, hidden_states, masked_token_indexes): - hidden_states = self.transform(hidden_states) - - if masked_token_indexes is not None: - hidden_states = torch.index_select( - hidden_states.view(-1, hidden_states.shape[-1]), 0, - masked_token_indexes) - - torch.cuda.nvtx.range_push( - "decoder input.size() = {}, weight.size() = {}".format( - hidden_states.size(), self.decoder.weight.size())) - hidden_states = self.decoder(hidden_states) + self.bias - torch.cuda.nvtx.range_pop() - return hidden_states - - -class BertOnlyMLMHead(nn.Module): - def __init__(self, config, bert_model_embedding_weights): - super(BertOnlyMLMHead, self).__init__() - self.predictions = BertLMPredictionHead(config, - bert_model_embedding_weights) - - def forward(self, sequence_output): - prediction_scores = self.predictions(sequence_output) - return prediction_scores - - -class BertOnlyNSPHead(nn.Module): - def __init__(self, config): - super(BertOnlyNSPHead, self).__init__() - self.seq_relationship = nn.Linear(config.hidden_size, 2) - - def forward(self, pooled_output): - seq_relationship_score = self.seq_relationship(pooled_output) - return seq_relationship_score - - -class BertPreTrainingHeads(nn.Module): - def __init__(self, config, bert_model_embedding_weights): - super(BertPreTrainingHeads, self).__init__() - self.predictions = BertLMPredictionHead(config, - bert_model_embedding_weights) - self.seq_relationship = nn.Linear(config.hidden_size, 2) - - def forward(self, - sequence_output, - pooled_output, - masked_token_indexes=None): - prediction_scores = self.predictions(sequence_output, - masked_token_indexes) - seq_relationship_score = self.seq_relationship(pooled_output) - return prediction_scores, seq_relationship_score - - -class BertPreTrainedModel(nn.Module): - """ An abstract class to handle weights initialization and - a simple interface for dowloading and loading pretrained models. - """ - def __init__(self, config, *inputs, **kwargs): - super(BertPreTrainedModel, self).__init__() - if not isinstance(config, BertConfig): - raise ValueError( - "Parameter config in `{}(config)` should be an instance of class `BertConfig`. " - "To create a model from a Google pretrained model use " - "`model = {}.from_pretrained(PRETRAINED_MODEL_NAME)`".format( - self.__class__.__name__, self.__class__.__name__)) - self.config = config - - def init_bert_weights(self, module): - """ Initialize the weights. - """ - if isinstance(module, (nn.Linear, nn.Embedding)): - # Slightly different from the TF version which uses truncated_normal for initialization - # cf https://github.com/pytorch/pytorch/pull/5617 - num_layers = self.config.num_hidden_layers - std = self.config.initializer_range - if hasattr(module, 'bert_output_layer'): - # "Accounting for accumulation on the residual path" - #print("Accounting for accumulation on the residual path") - std = self.config.initializer_range / math.sqrt( - 2.0 * num_layers) - module.weight.data.normal_(mean=0.0, std=std) - elif isinstance(module, BertLayerNorm): - module.bias.data.zero_() - module.weight.data.fill_(1.0) - if isinstance(module, nn.Linear) and module.bias is not None: - module.bias.data.zero_() - - @classmethod - def from_pretrained(cls, - pretrained_model_name_or_path, - state_dict=None, - cache_dir=None, - from_tf=False, - *inputs, - **kwargs): - """ - Instantiate a BertPreTrainedModel from a pre-trained model file or a pytorch state dict. - Download and cache the pre-trained model file if needed. - - Params: - pretrained_model_name_or_path: either: - - a str with the name of a pre-trained model to load selected in the list of: - . `bert-base-uncased` - . `bert-large-uncased` - . `bert-base-cased` - . `bert-large-cased` - . `bert-base-multilingual-uncased` - . `bert-base-multilingual-cased` - . `bert-base-chinese` - - a path or url to a pretrained model archive containing: - . `bert_config.json` a configuration file for the model - . `pytorch_model.bin` a PyTorch dump of a BertForPreTraining instance - - a path or url to a pretrained model archive containing: - . `bert_config.json` a configuration file for the model - . `model.chkpt` a TensorFlow checkpoint - from_tf: should we load the weights from a locally saved TensorFlow checkpoint - cache_dir: an optional path to a folder in which the pre-trained models will be cached. - state_dict: an optional state dictionnary (collections.OrderedDict object) to use instead of Google pre-trained models - *inputs, **kwargs: additional input for the specific Bert class - (ex: num_labels for BertForSequenceClassification) - """ - if pretrained_model_name_or_path in PRETRAINED_MODEL_ARCHIVE_MAP: - archive_file = PRETRAINED_MODEL_ARCHIVE_MAP[ - pretrained_model_name_or_path] - else: - archive_file = pretrained_model_name_or_path - # redirect to the cache, if necessary - try: - resolved_archive_file = cached_path(archive_file, - cache_dir=cache_dir) - except EnvironmentError: - logger.error( - "Model name '{}' was not found in model name list ({}). " - "We assumed '{}' was a path or url but couldn't find any file " - "associated to this path or url.".format( - pretrained_model_name_or_path, - ', '.join(PRETRAINED_MODEL_ARCHIVE_MAP.keys()), - archive_file)) - return None - if resolved_archive_file == archive_file: - logger.info("loading archive file {}".format(archive_file)) - else: - logger.info("loading archive file {} from cache at {}".format( - archive_file, resolved_archive_file)) - tempdir = None - if os.path.isdir(resolved_archive_file) or from_tf: - serialization_dir = resolved_archive_file - else: - # Extract archive to temp dir - tempdir = tempfile.mkdtemp() - logger.info("extracting archive file {} to temp dir {}".format( - resolved_archive_file, tempdir)) - with tarfile.open(resolved_archive_file, 'r:gz') as archive: - archive.extractall(tempdir) - serialization_dir = tempdir - # Load config - config_file = os.path.join(serialization_dir, CONFIG_NAME) - config = BertConfig.from_json_file(config_file) - logger.info("Model config {}".format(config)) - # Instantiate model. - model = cls(config, *inputs, **kwargs) - if state_dict is None and not from_tf: - weights_path = os.path.join(serialization_dir, WEIGHTS_NAME) - state_dict = torch.load( - weights_path, - map_location='cpu' if not torch.cuda.is_available() else None) - if tempdir: - # Clean up temp dir - shutil.rmtree(tempdir) - if from_tf: - # Directly load from a TensorFlow checkpoint - weights_path = os.path.join(serialization_dir, TF_WEIGHTS_NAME) - return load_tf_weights_in_bert(model, weights_path) - # Load from a PyTorch state_dict - old_keys = [] - new_keys = [] - for key in state_dict.keys(): - new_key = None - if 'gamma' in key: - new_key = key.replace('gamma', 'weight') - if 'beta' in key: - new_key = key.replace('beta', 'bias') - if new_key: - old_keys.append(key) - new_keys.append(new_key) - for old_key, new_key in zip(old_keys, new_keys): - state_dict[new_key] = state_dict.pop(old_key) - - missing_keys = [] - unexpected_keys = [] - error_msgs = [] - # copy state_dict so _load_from_state_dict can modify it - metadata = getattr(state_dict, '_metadata', None) - state_dict = state_dict.copy() - if metadata is not None: - state_dict._metadata = metadata - - def load(module, prefix=''): - local_metadata = {} if metadata is None else metadata.get( - prefix[:-1], {}) - module._load_from_state_dict(state_dict, prefix, local_metadata, - True, missing_keys, unexpected_keys, - error_msgs) - for name, child in module._modules.items(): - if child is not None: - load(child, prefix + name + '.') - - start_prefix = '' - if not hasattr(model, 'bert') and any( - s.startswith('bert.') for s in state_dict.keys()): - start_prefix = 'bert.' - load(model, prefix=start_prefix) - if len(missing_keys) > 0: - logger.info( - "Weights of {} not initialized from pretrained model: {}". - format(model.__class__.__name__, missing_keys)) - if len(unexpected_keys) > 0: - logger.info( - "Weights from pretrained model not used in {}: {}".format( - model.__class__.__name__, unexpected_keys)) - if len(error_msgs) > 0: - raise RuntimeError( - 'Error(s) in loading state_dict for {}:\n\t{}'.format( - model.__class__.__name__, "\n\t".join(error_msgs))) - return model - - -class BertModel(BertPreTrainedModel): - """BERT model ("Bidirectional Embedding Representations from a Transformer"). - - Params: - config: a BertConfig class instance with the configuration to build a new model - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `output_all_encoded_layers`: boolean which controls the content of the `encoded_layers` output as described below. Default: `True`. - - Outputs: Tuple of (encoded_layers, pooled_output) - `encoded_layers`: controled by `output_all_encoded_layers` argument: - - `output_all_encoded_layers=True`: outputs a list of the full sequences of encoded-hidden-states at the end - of each attention block (i.e. 12 full sequences for BERT-base, 24 for BERT-large), each - encoded-hidden-state is a torch.FloatTensor of size [batch_size, sequence_length, hidden_size], - - `output_all_encoded_layers=False`: outputs only the full sequence of hidden-states corresponding - to the last attention block of shape [batch_size, sequence_length, hidden_size], - `pooled_output`: a torch.FloatTensor of size [batch_size, hidden_size] which is the output of a - classifier pretrained on top of the hidden state associated to the first character of the - input (`CLS`) to train on the Next-Sentence task (see BERT's paper). - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = modeling.BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - model = modeling.BertModel(config=config) - all_encoder_layers, pooled_output = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config, args=None): - super(BertModel, self).__init__(config) - self.embeddings = BertEmbeddings(config) - # set pad_token_id that is used for sparse attention padding - self.pad_token_id = config.pad_token_id if hasattr( - config, 'pad_token_id') and config.pad_token_id is not None else 0 - # set sparse_attention_config if it has been selected - self.sparse_attention_config = get_sparse_attention_config( - args, config.num_attention_heads) - self.sparse_attention_utils = get_sparse_attention_utils( - self.sparse_attention_config) - self.encoder = BertEncoder( - config, args, sparse_attention_config=self.sparse_attention_config) - self.pooler = BertPooler(config) - self.apply(self.init_bert_weights) - logger.info("Init BERT pretrain model") - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - output_all_encoded_layers=True, - checkpoint_activations=False, - progressive_layer_drop=False, - theta=0.5): - if attention_mask is None: - attention_mask = torch.ones_like(input_ids) - if token_type_ids is None: - token_type_ids = torch.zeros_like(input_ids) - - # We create a 3D attention mask from a 2D tensor mask. - # Sizes are [batch_size, 1, 1, to_seq_length] - # So we can broadcast to [batch_size, num_heads, from_seq_length, to_seq_length] - # this attention mask is more simple than the triangular masking of causal attention - # used in OpenAI GPT, we just need to prepare the broadcast dimension here. - extended_attention_mask = attention_mask.unsqueeze(1).unsqueeze(2) - - # Since attention_mask is 1.0 for positions we want to attend and 0.0 for - # masked positions, this operation will create a tensor which is 0.0 for - # positions we want to attend and -10000.0 for masked positions. - # Since we are adding it to the raw scores before the softmax, this is - # effectively the same as removing these entirely. - extended_attention_mask = extended_attention_mask.to( - dtype=next(self.parameters()).dtype) # fp16 compatibility - extended_attention_mask = (1.0 - extended_attention_mask) * -10000.0 - - # If BertEncoder uses sparse attention, it needs to be padded based on the sparse attention block size - if self.sparse_attention_config is not None: - pad_len, input_ids, attention_mask, token_type_ids, position_ids, inputs_embeds = self.sparse_attention_utils.pad_to_block_size( - block_size=self.sparse_attention_config.block, - input_ids=input_ids, - attention_mask=extended_attention_mask, - token_type_ids=token_type_ids, - position_ids=None, - inputs_embeds=None, - pad_token_id=self.pad_token_id, - model_mbeddings=self.embeddings) - - embedding_output = self.embeddings(input_ids, token_type_ids) - encoded_layers = self.encoder( - embedding_output, - extended_attention_mask, - output_all_encoded_layers=output_all_encoded_layers, - checkpoint_activations=checkpoint_activations, - progressive_layer_drop=progressive_layer_drop, - theta=theta) - sequence_output = encoded_layers[-1] - pooled_output = self.pooler(sequence_output) - - # If BertEncoder uses sparse attention, and input_ids were padded, sequence output needs to be unpadded to original length - if self.sparse_attention_config is not None and pad_len > 0: - encoded_layers[ - -1] = self.sparse_attention_utils.unpad_sequence_output( - pad_len, encoded_layers[-1]) - - if not output_all_encoded_layers: - encoded_layers = encoded_layers[-1] - return encoded_layers, pooled_output - - -class BertForPreTrainingPreLN(BertPreTrainedModel): - """BERT model with pre-training heads. - This module comprises the BERT model followed by the two pre-training heads: - - the masked language modeling head, and - - the next sentence classification head. - - Params: - config: a BertConfig class instance with the configuration to build a new model. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `masked_lm_labels`: optional masked language modeling labels: torch.LongTensor of shape [batch_size, sequence_length] - with indices selected in [-1, 0, ..., vocab_size]. All labels set to -1 are ignored (masked), the loss - is only computed for the labels set in [0, ..., vocab_size] - `next_sentence_label`: optional next sentence classification loss: torch.LongTensor of shape [batch_size] - with indices selected in [0, 1]. - 0 => next sentence is the continuation, 1 => next sentence is a random sentence. - - Outputs: - if `masked_lm_labels` and `next_sentence_label` are not `None`: - Outputs the total_loss which is the sum of the masked language modeling loss and the next - sentence classification loss. - if `masked_lm_labels` or `next_sentence_label` is `None`: - Outputs a tuple comprising - - the masked language modeling logits of shape [batch_size, sequence_length, vocab_size], and - - the next sentence classification logits of shape [batch_size, 2]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - model = BertForPreTraining(config) - masked_lm_logits_scores, seq_relationship_logits = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config, args): - super(BertForPreTrainingPreLN, self).__init__(config) - self.bert = BertModel(config, args) - self.cls = BertPreTrainingHeads( - config, self.bert.embeddings.word_embeddings.weight) - self.apply(self.init_bert_weights) - self.args = args - - def forward(self, batch, **kwargs): - progressive_layer_drop = kwargs.get('progressive_layer_drop', False) - theta = kwargs.get('pld_theta', 1.0) - - input_ids = batch[1] - token_type_ids = batch[3] - attention_mask = batch[2] - masked_lm_labels = batch[5] - next_sentence_label = batch[4] - checkpoint_activations = False - - sequence_output, pooled_output = self.bert( - input_ids, - token_type_ids, - attention_mask, - output_all_encoded_layers=False, - checkpoint_activations=checkpoint_activations, - progressive_layer_drop=progressive_layer_drop, - theta=theta) - - if masked_lm_labels is not None and next_sentence_label is not None: - # filter out all masked labels. - masked_token_indexes = torch.nonzero( - (masked_lm_labels + 1).view(-1)).view(-1) - prediction_scores, seq_relationship_score = self.cls( - sequence_output, pooled_output, masked_token_indexes) - target = torch.index_select(masked_lm_labels.view(-1), 0, - masked_token_indexes) - - loss_fct = CrossEntropyLoss(ignore_index=-1) - masked_lm_loss = loss_fct( - prediction_scores.view(-1, self.config.vocab_size), target) - next_sentence_loss = loss_fct(seq_relationship_score.view(-1, 2), - next_sentence_label.view(-1)) - total_loss = masked_lm_loss + next_sentence_loss - return total_loss - else: - prediction_scores, seq_relationship_score = self.cls( - sequence_output, pooled_output) - return prediction_scores, seq_relationship_score - - -class BertForMaskedLM(BertPreTrainedModel): - """BERT model with the masked language modeling head. - This module comprises the BERT model followed by the masked language modeling head. - - Params: - config: a BertConfig class instance with the configuration to build a new model. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `masked_lm_labels`: masked language modeling labels: torch.LongTensor of shape [batch_size, sequence_length] - with indices selected in [-1, 0, ..., vocab_size]. All labels set to -1 are ignored (masked), the loss - is only computed for the labels set in [0, ..., vocab_size] - - Outputs: - if `masked_lm_labels` is not `None`: - Outputs the masked language modeling loss. - if `masked_lm_labels` is `None`: - Outputs the masked language modeling logits of shape [batch_size, sequence_length, vocab_size]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - model = BertForMaskedLM(config) - masked_lm_logits_scores = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config): - super(BertForMaskedLM, self).__init__(config) - self.bert = BertModel(config) - self.cls = BertOnlyMLMHead(config, - self.bert.embeddings.word_embeddings.weight) - self.apply(self.init_bert_weights) - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - masked_lm_labels=None, - checkpoint_activations=False): - sequence_output, _ = self.bert(input_ids, - token_type_ids, - attention_mask, - output_all_encoded_layers=False) - prediction_scores = self.cls(sequence_output) - - if masked_lm_labels is not None: - loss_fct = CrossEntropyLoss(ignore_index=-1) - masked_lm_loss = loss_fct( - prediction_scores.view(-1, self.config.vocab_size), - masked_lm_labels.view(-1)) - return masked_lm_loss - else: - return prediction_scores - - -class BertForNextSentencePrediction(BertPreTrainedModel): - """BERT model with next sentence prediction head. - This module comprises the BERT model followed by the next sentence classification head. - - Params: - config: a BertConfig class instance with the configuration to build a new model. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `next_sentence_label`: next sentence classification loss: torch.LongTensor of shape [batch_size] - with indices selected in [0, 1]. - 0 => next sentence is the continuation, 1 => next sentence is a random sentence. - - Outputs: - if `next_sentence_label` is not `None`: - Outputs the total_loss which is the sum of the masked language modeling loss and the next - sentence classification loss. - if `next_sentence_label` is `None`: - Outputs the next sentence classification logits of shape [batch_size, 2]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - model = BertForNextSentencePrediction(config) - seq_relationship_logits = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config): - super(BertForNextSentencePrediction, self).__init__(config) - self.bert = BertModel(config) - self.cls = BertOnlyNSPHead(config) - self.apply(self.init_bert_weights) - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - next_sentence_label=None, - checkpoint_activations=False): - _, pooled_output = self.bert(input_ids, - token_type_ids, - attention_mask, - output_all_encoded_layers=False) - seq_relationship_score = self.cls(pooled_output) - - if next_sentence_label is not None: - loss_fct = CrossEntropyLoss(ignore_index=-1) - next_sentence_loss = loss_fct(seq_relationship_score.view(-1, 2), - next_sentence_label.view(-1)) - return next_sentence_loss - else: - return seq_relationship_score - - -class BertForSequenceClassification(BertPreTrainedModel): - """BERT model for classification. - This module is composed of the BERT model with a linear layer on top of - the pooled output. - - Params: - `config`: a BertConfig class instance with the configuration to build a new model. - `num_labels`: the number of classes for the classifier. Default = 2. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `labels`: labels for the classification output: torch.LongTensor of shape [batch_size] - with indices selected in [0, ..., num_labels]. - - Outputs: - if `labels` is not `None`: - Outputs the CrossEntropy classification loss of the output with the labels. - if `labels` is `None`: - Outputs the classification logits of shape [batch_size, num_labels]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - num_labels = 2 - - model = BertForSequenceClassification(config, num_labels) - logits = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, args, config, num_labels): - super(BertForSequenceClassification, self).__init__(config) - self.num_labels = num_labels - self.bert = BertModel(config, args) - self.dropout = nn.Dropout(config.hidden_dropout_prob) - self.classifier = nn.Linear(config.hidden_size, num_labels) - self.apply(self.init_bert_weights) - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - labels=None, - checkpoint_activations=False): - _, pooled_output = self.bert(input_ids, - token_type_ids, - attention_mask, - output_all_encoded_layers=False) - pooled_output = self.dropout(pooled_output) - logits = self.classifier(pooled_output) - - if labels is not None: - loss_fct = CrossEntropyLoss() - loss = loss_fct(logits.view(-1, self.num_labels), labels.view(-1)) - return loss - else: - return logits - - -class BertForMultipleChoice(BertPreTrainedModel): - """BERT model for multiple choice tasks. - This module is composed of the BERT model with a linear layer on top of - the pooled output. - - Params: - `config`: a BertConfig class instance with the configuration to build a new model. - `num_choices`: the number of classes for the classifier. Default = 2. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, num_choices, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, num_choices, sequence_length] - with the token types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` - and type 1 corresponds to a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, num_choices, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `labels`: labels for the classification output: torch.LongTensor of shape [batch_size] - with indices selected in [0, ..., num_choices]. - - Outputs: - if `labels` is not `None`: - Outputs the CrossEntropy classification loss of the output with the labels. - if `labels` is `None`: - Outputs the classification logits of shape [batch_size, num_labels]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[[31, 51, 99], [15, 5, 0]], [[12, 16, 42], [14, 28, 57]]]) - input_mask = torch.LongTensor([[[1, 1, 1], [1, 1, 0]],[[1,1,0], [1, 0, 0]]]) - token_type_ids = torch.LongTensor([[[0, 0, 1], [0, 1, 0]],[[0, 1, 1], [0, 0, 1]]]) - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - num_choices = 2 - - model = BertForMultipleChoice(config, num_choices) - logits = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config, num_choices): - super(BertForMultipleChoice, self).__init__(config) - self.num_choices = num_choices - self.bert = BertModel(config) - self.dropout = nn.Dropout(config.hidden_dropout_prob) - self.classifier = nn.Linear(config.hidden_size, 1) - self.apply(self.init_bert_weights) - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - labels=None, - checkpoint_activations=False): - flat_input_ids = input_ids.view(-1, input_ids.size(-1)) - flat_token_type_ids = token_type_ids.view(-1, token_type_ids.size(-1)) - flat_attention_mask = attention_mask.view(-1, attention_mask.size(-1)) - _, pooled_output = self.bert(flat_input_ids, - flat_token_type_ids, - flat_attention_mask, - output_all_encoded_layers=False) - pooled_output = self.dropout(pooled_output) - logits = self.classifier(pooled_output) - reshaped_logits = logits.view(-1, self.num_choices) - - if labels is not None: - loss_fct = CrossEntropyLoss() - loss = loss_fct(reshaped_logits, labels) - return loss - else: - return reshaped_logits - - -class BertForTokenClassification(BertPreTrainedModel): - """BERT model for token-level classification. - This module is composed of the BERT model with a linear layer on top of - the full hidden state of the last layer. - - Params: - `config`: a BertConfig class instance with the configuration to build a new model. - `num_labels`: the number of classes for the classifier. Default = 2. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `labels`: labels for the classification output: torch.LongTensor of shape [batch_size, sequence_length] - with indices selected in [0, ..., num_labels]. - - Outputs: - if `labels` is not `None`: - Outputs the CrossEntropy classification loss of the output with the labels. - if `labels` is `None`: - Outputs the classification logits of shape [batch_size, sequence_length, num_labels]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - num_labels = 2 - - model = BertForTokenClassification(config, num_labels) - logits = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config, num_labels): - super(BertForTokenClassification, self).__init__(config) - self.num_labels = num_labels - self.bert = BertModel(config) - self.dropout = nn.Dropout(config.hidden_dropout_prob) - self.classifier = nn.Linear(config.hidden_size, num_labels) - self.apply(self.init_bert_weights) - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - labels=None, - checkpoint_activations=False): - sequence_output, _ = self.bert(input_ids, - token_type_ids, - attention_mask, - output_all_encoded_layers=False) - sequence_output = self.dropout(sequence_output) - logits = self.classifier(sequence_output) - - if labels is not None: - loss_fct = CrossEntropyLoss() - # Only keep active parts of the loss - if attention_mask is not None: - active_loss = attention_mask.view(-1) == 1 - active_logits = logits.view(-1, self.num_labels)[active_loss] - active_labels = labels.view(-1)[active_loss] - loss = loss_fct(active_logits, active_labels) - else: - loss = loss_fct(logits.view(-1, self.num_labels), - labels.view(-1)) - return loss - else: - return logits - - -class BertForQuestionAnswering(BertPreTrainedModel): - """BERT model for Question Answering (span extraction). - This module is composed of the BERT model with a linear layer on top of - the sequence output that computes start_logits and end_logits - - Params: - `config`: a BertConfig class instance with the configuration to build a new model. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `start_positions`: position of the first token for the labeled span: torch.LongTensor of shape [batch_size]. - Positions are clamped to the length of the sequence and position outside of the sequence are not taken - into account for computing the loss. - `end_positions`: position of the last token for the labeled span: torch.LongTensor of shape [batch_size]. - Positions are clamped to the length of the sequence and position outside of the sequence are not taken - into account for computing the loss. - - Outputs: - if `start_positions` and `end_positions` are not `None`: - Outputs the total_loss which is the sum of the CrossEntropy loss for the start and end token positions. - if `start_positions` or `end_positions` is `None`: - Outputs a tuple of start_logits, end_logits which are the logits respectively for the start and end - position tokens of shape [batch_size, sequence_length]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - model = BertForQuestionAnswering(config) - start_logits, end_logits = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config): - super(BertForQuestionAnswering, self).__init__(config) - self.bert = BertModel(config) - # TODO check with Google if it's normal there is no dropout on the token classifier of SQuAD in the TF version - # self.dropout = nn.Dropout(config.hidden_dropout_prob) - self.qa_outputs = nn.Linear(config.hidden_size, 2) - self.apply(self.init_bert_weights) - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - start_positions=None, - end_positions=None, - checkpoint_activations=False): - sequence_output, _ = self.bert(input_ids, - token_type_ids, - attention_mask, - output_all_encoded_layers=False) - logits = self.qa_outputs(sequence_output) - start_logits, end_logits = logits.split(1, dim=-1) - start_logits = start_logits.squeeze(-1) - end_logits = end_logits.squeeze(-1) - - if start_positions is not None and end_positions is not None: - # If we are on multi-GPU, split add a dimension - if len(start_positions.size()) > 1: - start_positions = start_positions.squeeze(-1) - if len(end_positions.size()) > 1: - end_positions = end_positions.squeeze(-1) - # sometimes the start/end positions are outside our model inputs, we ignore these terms - ignored_index = start_logits.size(1) - start_positions.clamp_(0, ignored_index) - end_positions.clamp_(0, ignored_index) - - loss_fct = CrossEntropyLoss(ignore_index=ignored_index) - start_loss = loss_fct(start_logits, start_positions) - end_loss = loss_fct(end_logits, end_positions) - total_loss = (start_loss + end_loss) / 2 - return total_loss - else: - return start_logits, end_logits diff --git a/training/BingBertGlue/run_glue_classifier_bert_base.py b/training/BingBertGlue/run_glue_classifier_bert_base.py index 08409ae60..64f5adf09 100755 --- a/training/BingBertGlue/run_glue_classifier_bert_base.py +++ b/training/BingBertGlue/run_glue_classifier_bert_base.py @@ -674,18 +674,10 @@ def main(): help="Whether to use Focal Loss for finetuning.") parser.add_argument('--gamma', type=float, default=0.5, help="Gamma parameter to be used in focal loss.") - parser.add_argument('--deepspeed_sparse_attention', - default=False, - action='store_true', - help='Use DeepSpeed sparse self attention.') parser.add_argument('--deepspeed_transformer_kernel', default=False, action='store_true', help='Use DeepSpeed transformer kernel to accelerate.') - parser.add_argument('--progressive_layer_drop', - default=False, - action='store_true', - help="Whether to enable progressive layer dropping or not") parser.add_argument( '--preln', action='store_true', @@ -807,10 +799,7 @@ def main(): "initializer_range": 0.02 } - if args.progressive_layer_drop: - print("BertBaseConfigPreLnLayerDrop") - from nvidia.modelingpreln_layerdrop import BertForSequenceClassification, BertConfig - elif args.preln: + if args.preln: from nvidia.modelingpreln import BertForSequenceClassification, BertConfig, BertLayer else: from nvidia.modeling import BertForSequenceClassification, BertConfig, BertLayer diff --git a/training/BingBertGlue/run_glue_classifier_bert_large.py b/training/BingBertGlue/run_glue_classifier_bert_large.py index 2f37d1c9c..57ec56bcd 100755 --- a/training/BingBertGlue/run_glue_classifier_bert_large.py +++ b/training/BingBertGlue/run_glue_classifier_bert_large.py @@ -747,10 +747,6 @@ def main(): type=float, default=0.5, help="Gamma parameter to be used in focal loss.") - parser.add_argument('--deepspeed_sparse_attention', - default=False, - action='store_true', - help='Use DeepSpeed sparse self attention.') parser.add_argument( '--preln', action='store_true', @@ -762,11 +758,6 @@ def main(): default=False, action='store_true', help='Use DeepSpeed transformer kernel to accelerate.') - parser.add_argument( - '--progressive_layer_drop', - default=False, - action='store_true', - help="Whether to enable progressive layer dropping or not") parser = deepspeed.add_config_arguments(parser) args = parser.parse_args() @@ -886,10 +877,7 @@ def main(): "initializer_range": 0.02 } - if args.progressive_layer_drop: - print("BertBaseConfigPreLnLayerDrop") - from nvidia.modelingpreln_layerdrop import BertForSequenceClassification, BertConfig, BertLayer - elif args.preln: + if args.preln: from nvidia.modelingpreln import BertForSequenceClassification, BertConfig, BertLayer else: from nvidia.modeling import BertForSequenceClassification, BertConfig, BertLayer diff --git a/training/BingBertGlue/turing/models.py b/training/BingBertGlue/turing/models.py index 35a8d202f..e9ec3e3f6 100755 --- a/training/BingBertGlue/turing/models.py +++ b/training/BingBertGlue/turing/models.py @@ -105,12 +105,7 @@ def __init__(self, args): self.config = args.config if not args.use_pretrain: - - if args.progressive_layer_drop: - print("BertConfigPreLnLayerDrop") - from nvidia.modelingpreln_layerdrop import BertForPreTrainingPreLN, BertConfig - else: - from nvidia.modelingpreln import BertForPreTrainingPreLN, BertConfig + from nvidia.modelingpreln import BertForPreTrainingPreLN, BertConfig bert_config = BertConfig(**self.config["bert_model_config"]) bert_config.vocab_size = len(args.tokenizer.vocab) diff --git a/training/BingBertSquad/1-bit_adam/mpi_ethernet/deepspeed_onebitadam_bsz96_config.json b/training/BingBertSquad/1-bit_adam/mpi_ethernet/deepspeed_onebitadam_bsz96_config.json deleted file mode 100644 index 9328dd2b7..000000000 --- a/training/BingBertSquad/1-bit_adam/mpi_ethernet/deepspeed_onebitadam_bsz96_config.json +++ /dev/null @@ -1,20 +0,0 @@ -{ - "train_batch_size": 96, - "train_micro_batch_size_per_gpu": 3, - "steps_per_print": 100, - "optimizer": { - "type": "OnebitAdam", - "params": { - "lr": 3e-5, - "freeze_step": 400, - "weight_decay": 0.0, - "bias_correction": false, - "cuda_aware": false, - "comm_backend_name": "mpi" - } - }, - "gradient_clipping": 1.0, - "fp16": { - "enabled": true - } -} diff --git a/training/BingBertSquad/1-bit_adam/mpi_ethernet/run_squad_deepspeed_onebitadam.sh b/training/BingBertSquad/1-bit_adam/mpi_ethernet/run_squad_deepspeed_onebitadam.sh deleted file mode 100644 index 5529fc775..000000000 --- a/training/BingBertSquad/1-bit_adam/mpi_ethernet/run_squad_deepspeed_onebitadam.sh +++ /dev/null @@ -1,60 +0,0 @@ -# If you are able to install pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs), -# we highly recommend you to use the NCCL-based 1-bit Adam -# which has better performance and ease of use -# (see scripts in DeepSpeedExamples/BingBertSquad/1-bit_adam/nccl -# and read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-adam/) - -NUM_NODES=4 -NGPU_PER_NODE=8 -MODEL_FILE="../../ckpt/bert-large-uncased-whole-word-masking-pytorch_model.bin" -ORIGIN_CONFIG_FILE="../../ckpt/bert-large-uncased-whole-word-masking-config.json" -SQUAD_DIR="../../data" -OUTPUT_DIR=$1 -LR=3e-5 -SEED=$RANDOM -MASTER_PORT=12345 -DROPOUT=0.1 - -sudo rm -rf ${OUTPUT_DIR} - -NGPU=$((NGPU_PER_NODE*NUM_NODES)) -EFFECTIVE_BATCH_SIZE=96 -MAX_GPU_BATCH_SIZE=3 -PER_GPU_BATCH_SIZE=$((EFFECTIVE_BATCH_SIZE/NGPU)) -if [[ $PER_GPU_BATCH_SIZE -lt $MAX_GPU_BATCH_SIZE ]]; then - GRAD_ACCUM_STEPS=1 -else - GRAD_ACCUM_STEPS=$((PER_GPU_BATCH_SIZE/MAX_GPU_BATCH_SIZE)) -fi -JOB_NAME="onebit_deepspeed_${NGPU}GPUs_${EFFECTIVE_BATCH_SIZE}batch_size" -config_json=deepspeed_onebitadam_bsz96_config.json - -# NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 are used to disable infiniband. Remove it if needed. -NCCL_TREE_THRESHOLD=0 NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 deepspeed --launcher=openmpi ../../nvidia_run_squad_deepspeed.py \ ---bert_model bert-large-uncased \ ---do_train \ ---do_lower_case \ ---predict_batch_size 3 \ ---do_predict \ ---train_file $SQUAD_DIR/train-v1.1.json \ ---predict_file $SQUAD_DIR/dev-v1.1.json \ ---train_batch_size $PER_GPU_BATCH_SIZE \ ---learning_rate ${LR} \ ---num_train_epochs 2.0 \ ---max_seq_length 384 \ ---doc_stride 128 \ ---output_dir $OUTPUT_DIR \ ---job_name ${JOB_NAME} \ ---gradient_accumulation_steps ${GRAD_ACCUM_STEPS} \ ---fp16 \ ---deepspeed \ ---deepspeed_mpi \ ---deepspeed_transformer_kernel \ ---deepspeed_config ${config_json} \ ---dropout ${DROPOUT} \ ---model_file $MODEL_FILE \ ---seed ${SEED} \ ---ckpt_type HF \ ---origin_bert_config_file ${ORIGIN_CONFIG_FILE} \ diff --git a/training/BingBertSquad/1-bit_adam/mpi_ethernet/run_squad_mpi_onebitadam.sh b/training/BingBertSquad/1-bit_adam/mpi_ethernet/run_squad_mpi_onebitadam.sh deleted file mode 100644 index f5b090643..000000000 --- a/training/BingBertSquad/1-bit_adam/mpi_ethernet/run_squad_mpi_onebitadam.sh +++ /dev/null @@ -1,60 +0,0 @@ -# If you are able to install pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs), -# we highly recommend you to use the NCCL-based 1-bit Adam -# which has better performance and ease of use -# (see scripts in DeepSpeedExamples/BingBertSquad/1-bit_adam/nccl -# and read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-adam/) - -NUM_NODES=4 -NGPU_PER_NODE=8 -MODEL_FILE="../../ckpt/bert-large-uncased-whole-word-masking-pytorch_model.bin" -ORIGIN_CONFIG_FILE="../../ckpt/bert-large-uncased-whole-word-masking-config.json" -SQUAD_DIR="../../data" -OUTPUT_DIR=$1 -LR=3e-5 -SEED=$RANDOM -MASTER_PORT=12345 -DROPOUT=0.1 - -sudo rm -rf ${OUTPUT_DIR} - -NGPU=$((NGPU_PER_NODE*NUM_NODES)) -EFFECTIVE_BATCH_SIZE=96 -MAX_GPU_BATCH_SIZE=3 -PER_GPU_BATCH_SIZE=$((EFFECTIVE_BATCH_SIZE/NGPU)) -if [[ $PER_GPU_BATCH_SIZE -lt $MAX_GPU_BATCH_SIZE ]]; then - GRAD_ACCUM_STEPS=1 -else - GRAD_ACCUM_STEPS=$((PER_GPU_BATCH_SIZE/MAX_GPU_BATCH_SIZE)) -fi -JOB_NAME="onebit_deepspeed_${NGPU}GPUs_${EFFECTIVE_BATCH_SIZE}batch_size" -config_json=deepspeed_onebitadam_bsz96_config.json - -# NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 are used to disable infiniband. Remove it if needed. -mpirun -n $NGPU -npernode $NGPU_PER_NODE -hostfile /job/hostfile -x UCX_TLS=tcp --mca btl ^openib --mca btl_tcp_if_include eth0 -x NCCL_TREE_THRESHOLD=0 -x NCCL_IB_DISABLE=1 -x NCCL_SOCKET_IFNAME=eth0 python ../../nvidia_run_squad_deepspeed.py \ ---bert_model bert-large-uncased \ ---do_train \ ---do_lower_case \ ---predict_batch_size 3 \ ---do_predict \ ---train_file $SQUAD_DIR/train-v1.1.json \ ---predict_file $SQUAD_DIR/dev-v1.1.json \ ---train_batch_size $PER_GPU_BATCH_SIZE \ ---learning_rate ${LR} \ ---num_train_epochs 2.0 \ ---max_seq_length 384 \ ---doc_stride 128 \ ---output_dir $OUTPUT_DIR \ ---job_name ${JOB_NAME} \ ---gradient_accumulation_steps ${GRAD_ACCUM_STEPS} \ ---fp16 \ ---deepspeed \ ---deepspeed_mpi \ ---deepspeed_transformer_kernel \ ---deepspeed_config ${config_json} \ ---dropout ${DROPOUT} \ ---model_file $MODEL_FILE \ ---seed ${SEED} \ ---ckpt_type HF \ ---origin_bert_config_file ${ORIGIN_CONFIG_FILE} \ diff --git a/training/BingBertSquad/1-bit_adam/mpi_infiniband/deepspeed_onebitadam_bsz96_config.json b/training/BingBertSquad/1-bit_adam/mpi_infiniband/deepspeed_onebitadam_bsz96_config.json deleted file mode 100755 index 4ba02582b..000000000 --- a/training/BingBertSquad/1-bit_adam/mpi_infiniband/deepspeed_onebitadam_bsz96_config.json +++ /dev/null @@ -1,20 +0,0 @@ -{ - "train_batch_size": 96, - "train_micro_batch_size_per_gpu": 3, - "steps_per_print": 100, - "optimizer": { - "type": "OnebitAdam", - "params": { - "lr": 3e-5, - "freeze_step": 400, - "weight_decay": 0.0, - "bias_correction": false, - "cuda_aware": true, - "comm_backend_name": "mpi" - } - }, - "gradient_clipping": 1.0, - "fp16": { - "enabled": true - } -} diff --git a/training/BingBertSquad/1-bit_adam/mpi_infiniband/run_squad_deepspeed_onebitadam.sh b/training/BingBertSquad/1-bit_adam/mpi_infiniband/run_squad_deepspeed_onebitadam.sh deleted file mode 100755 index 751c65570..000000000 --- a/training/BingBertSquad/1-bit_adam/mpi_infiniband/run_squad_deepspeed_onebitadam.sh +++ /dev/null @@ -1,59 +0,0 @@ -# If you are able to install pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs), -# we highly recommend you to use the NCCL-based 1-bit Adam -# which has better performance and ease of use -# (see scripts in DeepSpeedExamples/BingBertSquad/1-bit_adam/nccl -# and read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-adam/) - -NUM_NODES=4 -NGPU_PER_NODE=8 -MODEL_FILE="../../ckpt/bert-large-uncased-whole-word-masking-pytorch_model.bin" -ORIGIN_CONFIG_FILE="../../ckpt/bert-large-uncased-whole-word-masking-config.json" -SQUAD_DIR="../../data" -OUTPUT_DIR=$1 -LR=3e-5 -SEED=$RANDOM -MASTER_PORT=12345 -DROPOUT=0.1 - -sudo rm -rf ${OUTPUT_DIR} - -NGPU=$((NGPU_PER_NODE*NUM_NODES)) -EFFECTIVE_BATCH_SIZE=96 -MAX_GPU_BATCH_SIZE=3 -PER_GPU_BATCH_SIZE=$((EFFECTIVE_BATCH_SIZE/NGPU)) -if [[ $PER_GPU_BATCH_SIZE -lt $MAX_GPU_BATCH_SIZE ]]; then - GRAD_ACCUM_STEPS=1 -else - GRAD_ACCUM_STEPS=$((PER_GPU_BATCH_SIZE/MAX_GPU_BATCH_SIZE)) -fi -JOB_NAME="onebit_deepspeed_${NGPU}GPUs_${EFFECTIVE_BATCH_SIZE}batch_size" -config_json=deepspeed_onebitadam_bsz96_config.json - -NCCL_TREE_THRESHOLD=0 deepspeed --launcher=mvapich ../../nvidia_run_squad_deepspeed.py \ ---bert_model bert-large-uncased \ ---do_train \ ---do_lower_case \ ---predict_batch_size 3 \ ---do_predict \ ---train_file $SQUAD_DIR/train-v1.1.json \ ---predict_file $SQUAD_DIR/dev-v1.1.json \ ---train_batch_size $PER_GPU_BATCH_SIZE \ ---learning_rate ${LR} \ ---num_train_epochs 2.0 \ ---max_seq_length 384 \ ---doc_stride 128 \ ---output_dir $OUTPUT_DIR \ ---job_name ${JOB_NAME} \ ---gradient_accumulation_steps ${GRAD_ACCUM_STEPS} \ ---fp16 \ ---deepspeed \ ---deepspeed_mpi \ ---deepspeed_transformer_kernel \ ---deepspeed_config ${config_json} \ ---dropout ${DROPOUT} \ ---model_file $MODEL_FILE \ ---seed ${SEED} \ ---ckpt_type HF \ ---origin_bert_config_file ${ORIGIN_CONFIG_FILE} \ diff --git a/training/BingBertSquad/1-bit_adam/mpi_infiniband/run_squad_mpi_onebitadam.sh b/training/BingBertSquad/1-bit_adam/mpi_infiniband/run_squad_mpi_onebitadam.sh deleted file mode 100755 index f02c7007e..000000000 --- a/training/BingBertSquad/1-bit_adam/mpi_infiniband/run_squad_mpi_onebitadam.sh +++ /dev/null @@ -1,59 +0,0 @@ -# If you are able to install pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs), -# we highly recommend you to use the NCCL-based 1-bit Adam -# which has better performance and ease of use -# (see scripts in DeepSpeedExamples/BingBertSquad/1-bit_adam/nccl -# and read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-adam/) - -NUM_NODES=4 -NGPU_PER_NODE=8 -MODEL_FILE="../../ckpt/bert-large-uncased-whole-word-masking-pytorch_model.bin" -ORIGIN_CONFIG_FILE="../../ckpt/bert-large-uncased-whole-word-masking-config.json" -SQUAD_DIR="../../data" -OUTPUT_DIR=$1 -LR=3e-5 -SEED=$RANDOM -MASTER_PORT=12345 -DROPOUT=0.1 - -sudo rm -rf ${OUTPUT_DIR} - -NGPU=$((NGPU_PER_NODE*NUM_NODES)) -EFFECTIVE_BATCH_SIZE=96 -MAX_GPU_BATCH_SIZE=3 -PER_GPU_BATCH_SIZE=$((EFFECTIVE_BATCH_SIZE/NGPU)) -if [[ $PER_GPU_BATCH_SIZE -lt $MAX_GPU_BATCH_SIZE ]]; then - GRAD_ACCUM_STEPS=1 -else - GRAD_ACCUM_STEPS=$((PER_GPU_BATCH_SIZE/MAX_GPU_BATCH_SIZE)) -fi -JOB_NAME="onebit_deepspeed_${NGPU}GPUs_${EFFECTIVE_BATCH_SIZE}batch_size" -config_json=deepspeed_onebitadam_bsz96_config.json - -mpirun -n $NGPU -ppn $NGPU_PER_NODE -f /tmp/deepspeed_mvapich_hostfile -env MV2_SUPPORT_DL=1 -env MV2_USE_GDR=0 -env MV2_USE_CUDA=1 -env MV2_USE_GDRCOPY=0 -env MV2_SMP_USE_CMA=0 -env MV2_DEBUG_SHOW_BACKTRACE=1 python ../../nvidia_run_squad_deepspeed.py \ ---bert_model bert-large-uncased \ ---do_train \ ---do_lower_case \ ---predict_batch_size 3 \ ---do_predict \ ---train_file $SQUAD_DIR/train-v1.1.json \ ---predict_file $SQUAD_DIR/dev-v1.1.json \ ---train_batch_size $PER_GPU_BATCH_SIZE \ ---learning_rate ${LR} \ ---num_train_epochs 2.0 \ ---max_seq_length 384 \ ---doc_stride 128 \ ---output_dir $OUTPUT_DIR \ ---job_name ${JOB_NAME} \ ---gradient_accumulation_steps ${GRAD_ACCUM_STEPS} \ ---fp16 \ ---deepspeed \ ---deepspeed_mpi \ ---deepspeed_transformer_kernel \ ---deepspeed_config ${config_json} \ ---dropout ${DROPOUT} \ ---model_file $MODEL_FILE \ ---seed ${SEED} \ ---ckpt_type HF \ ---origin_bert_config_file ${ORIGIN_CONFIG_FILE} \ diff --git a/training/BingBertSquad/1-bit_adam/nccl/deepspeed_onebitadam_bsz96_config.json b/training/BingBertSquad/1-bit_adam/nccl/deepspeed_onebitadam_bsz96_config.json deleted file mode 100644 index 56d3e55cb..000000000 --- a/training/BingBertSquad/1-bit_adam/nccl/deepspeed_onebitadam_bsz96_config.json +++ /dev/null @@ -1,20 +0,0 @@ -{ - "train_batch_size": 96, - "train_micro_batch_size_per_gpu": 3, - "steps_per_print": 100, - "optimizer": { - "type": "OnebitAdam", - "params": { - "lr": 3e-5, - "freeze_step": 400, - "weight_decay": 0.0, - "bias_correction": false, - "cuda_aware": false, - "comm_backend_name": "nccl" - } - }, - "gradient_clipping": 1.0, - "fp16": { - "enabled": true - } -} diff --git a/training/BingBertSquad/1-bit_adam/nccl/run_squad_deepspeed_onebitadam.sh b/training/BingBertSquad/1-bit_adam/nccl/run_squad_deepspeed_onebitadam.sh deleted file mode 100644 index 3c9b3dc39..000000000 --- a/training/BingBertSquad/1-bit_adam/nccl/run_squad_deepspeed_onebitadam.sh +++ /dev/null @@ -1,56 +0,0 @@ -# This script requires pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs). -# Read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-adam/ - -NUM_NODES=4 -NGPU_PER_NODE=8 -MODEL_FILE="../../ckpt/bert-large-uncased-whole-word-masking-pytorch_model.bin" -ORIGIN_CONFIG_FILE="../../ckpt/bert-large-uncased-whole-word-masking-config.json" -SQUAD_DIR="../../data" -OUTPUT_DIR=$1 -LR=3e-5 -SEED=$RANDOM -MASTER_PORT=12345 -DROPOUT=0.1 - -sudo rm -rf ${OUTPUT_DIR} - -NGPU=$((NGPU_PER_NODE*NUM_NODES)) -EFFECTIVE_BATCH_SIZE=96 -MAX_GPU_BATCH_SIZE=3 -PER_GPU_BATCH_SIZE=$((EFFECTIVE_BATCH_SIZE/NGPU)) -if [[ $PER_GPU_BATCH_SIZE -lt $MAX_GPU_BATCH_SIZE ]]; then - GRAD_ACCUM_STEPS=1 -else - GRAD_ACCUM_STEPS=$((PER_GPU_BATCH_SIZE/MAX_GPU_BATCH_SIZE)) -fi -JOB_NAME="onebit_deepspeed_${NGPU}GPUs_${EFFECTIVE_BATCH_SIZE}batch_size" -config_json=deepspeed_onebitadam_bsz96_config.json - -# NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 are used to disable infiniband. Remove it if needed. -NCCL_TREE_THRESHOLD=0 NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 deepspeed ../../nvidia_run_squad_deepspeed.py \ ---bert_model bert-large-uncased \ ---do_train \ ---do_lower_case \ ---predict_batch_size 3 \ ---do_predict \ ---train_file $SQUAD_DIR/train-v1.1.json \ ---predict_file $SQUAD_DIR/dev-v1.1.json \ ---train_batch_size $PER_GPU_BATCH_SIZE \ ---learning_rate ${LR} \ ---num_train_epochs 2.0 \ ---max_seq_length 384 \ ---doc_stride 128 \ ---output_dir $OUTPUT_DIR \ ---job_name ${JOB_NAME} \ ---gradient_accumulation_steps ${GRAD_ACCUM_STEPS} \ ---fp16 \ ---deepspeed \ ---deepspeed_transformer_kernel \ ---deepspeed_config ${config_json} \ ---dropout ${DROPOUT} \ ---model_file $MODEL_FILE \ ---seed ${SEED} \ ---ckpt_type HF \ ---origin_bert_config_file ${ORIGIN_CONFIG_FILE} \ diff --git a/training/BingBertSquad/turing/modelingpreln_layerdrop.py b/training/BingBertSquad/turing/modelingpreln_layerdrop.py deleted file mode 100755 index 4224cf208..000000000 --- a/training/BingBertSquad/turing/modelingpreln_layerdrop.py +++ /dev/null @@ -1,1652 +0,0 @@ -# DeepSpeed note, code taken from commit 3d59216cec89a363649b4fe3d15295ba936ced0f -# https://github.com/NVIDIA/DeepLearningExamples/blob/master/PyTorch/LanguageModeling/BERT/modeling.py - -# coding=utf-8 -# Copyright 2018 The Google AI Language Team Authors and The HugginFace Inc. team. -# Copyright (c) 2018, NVIDIA CORPORATION. All rights reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -"""PyTorch BERT model.""" - -from __future__ import absolute_import, division, print_function, unicode_literals - -import copy -import json -import logging -import math -import os -import shutil -import tarfile -import tempfile -import sys -from io import open - -import torch -from torch import nn -from torch.nn import CrossEntropyLoss -from torch.utils import checkpoint - -from turing.file_utils import cached_path - -from torch.nn import Module -from torch.nn.parameter import Parameter -import torch.nn.functional as F -import torch.nn.init as init - -import numpy as np - -logger = logging.getLogger(__name__) - -PRETRAINED_MODEL_ARCHIVE_MAP = { - 'bert-base-uncased': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-base-uncased.tar.gz", - 'bert-large-uncased': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-large-uncased.tar.gz", - 'bert-base-cased': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-base-cased.tar.gz", - 'bert-large-cased': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-large-cased.tar.gz", - 'bert-base-multilingual-uncased': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-base-multilingual-uncased.tar.gz", - 'bert-base-multilingual-cased': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-base-multilingual-cased.tar.gz", - 'bert-base-chinese': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-base-chinese.tar.gz", -} -CONFIG_NAME = 'bert_config.json' -WEIGHTS_NAME = 'pytorch_model.bin' -TF_WEIGHTS_NAME = 'model.ckpt' - - -def get_deepspeed_config(args): - if hasattr(args, 'deepspeed_config') and args.deepspeed_config: - from deepspeed import DeepSpeedConfig - return DeepSpeedConfig(args.deepspeed_config) - else: - raise RuntimeError('deepspeed_config is not found in args.') - - -def get_sparse_attention_config(args, num_heads): - if args.deepspeed_sparse_attention: - ds_config = get_deepspeed_config(args) - if hasattr(ds_config, - 'sparse_attention') and ds_config.sparse_attention: - sa_config = ds_config.sparse_attention - sa_mode = sa_config.get('mode') - if (sa_mode == 'dense'): - from deepspeed.ops.sparse_attention import DenseSparsityConfig as STConfig - elif (sa_mode == 'fixed'): - from deepspeed.ops.sparse_attention import FixedSparsityConfig as STConfig - elif (sa_mode == 'bigbird'): - from deepspeed.ops.sparse_attention import BigBirdSparsityConfig as STConfig - elif (sa_mode == 'bslongformer'): - from deepspeed.ops.sparse_attention import BSLongformerSparsityConfig as STConfig - elif (sa_mode == 'variable'): - from deepspeed.ops.sparse_attention import VariableSparsityConfig as STConfig - else: - raise NotImplementedError( - f'Given sparsity mode, {sa_mode}, has not been implemented yet!' - ) - del sa_config['mode'] - return STConfig(num_heads=num_heads, **sa_config) - else: - from deepspeed.ops.sparse_attention import FixedSparsityConfig as STConfig - print( - 'deepspeed sparse attention is not set; Fixed sparsity is used as default.' - ) - return STConfig(num_heads=num_heads) - else: - return None - -def get_sparse_attention_utils(sparse_attention_config): - if sparse_attention_config is not None: - from deepspeed.ops.sparse_attention import SparseAttentionUtils - return SparseAttentionUtils - return None - -def load_tf_weights_in_bert(model, tf_checkpoint_path): - """ Load tf checkpoints in a pytorch model - """ - try: - import re - import numpy as np - import tensorflow as tf - except ImportError: - print( - "Loading a TensorFlow models in PyTorch, requires TensorFlow to be installed. Please see " - "https://www.tensorflow.org/install/ for installation instructions." - ) - raise - tf_path = os.path.abspath(tf_checkpoint_path) - print("Converting TensorFlow checkpoint from {}".format(tf_path)) - # Load weights from TF model - init_vars = tf.train.list_variables(tf_path) - names = [] - arrays = [] - for name, shape in init_vars: - print("Loading TF weight {} with shape {}".format(name, shape)) - array = tf.train.load_variable(tf_path, name) - names.append(name) - arrays.append(array) - - for name, array in zip(names, arrays): - name = name.split('/') - # adam_v and adam_m are variables used in AdamWeightDecayOptimizer to calculated m and v - # which are not required for using pretrained model - if any(n in ["adam_v", "adam_m"] for n in name): - print("Skipping {}".format("/".join(name))) - continue - pointer = model - for m_name in name: - if re.fullmatch(r'[A-Za-z]+_\d+', m_name): - l = re.split(r'_(\d+)', m_name) - else: - l = [m_name] - if l[0] == 'kernel' or l[0] == 'gamma': - pointer = getattr(pointer, 'weight') - elif l[0] == 'output_bias' or l[0] == 'beta': - pointer = getattr(pointer, 'bias') - elif l[0] == 'output_weights': - pointer = getattr(pointer, 'weight') - else: - pointer = getattr(pointer, l[0]) - if len(l) >= 2: - num = int(l[1]) - pointer = pointer[num] - if m_name[-11:] == '_embeddings': - pointer = getattr(pointer, 'weight') - elif m_name == 'kernel': - array = np.transpose(array) - try: - assert pointer.shape == array.shape - except AssertionError as e: - e.args += (pointer.shape, array.shape) - raise - print("Initialize PyTorch weight {}".format(name)) - pointer.data = torch.from_numpy(array) - return model - - -@torch.jit.script -def f_gelu(x): - pdtype = x.dtype - x = x.float() - y = x * 0.5 * (1.0 + torch.erf(x / math.sqrt(2.0))) - return y.to(pdtype) - - -@torch.jit.script -def bias_gelu(bias, y): - x = bias + y - return x * 0.5 * (1.0 + torch.erf(x / 1.41421)) - - -@torch.jit.script -def bias_tanh(bias, y): - x = bias + y - return torch.tanh(x) - - -def gelu(x): - """Implementation of the gelu activation function. - For information: OpenAI GPT's gelu is slightly different (and gives slightly different results): - 0.5 * x * (1 + torch.tanh(math.sqrt(2 / math.pi) * (x + 0.044715 * torch.pow(x, 3)))) - Also see https://arxiv.org/abs/1606.08415 - """ - return f_gelu(x) - - -def swish(x): - return x * torch.sigmoid(x) - - -ACT2FN = {"gelu": gelu, "relu": torch.nn.functional.relu, "swish": swish} - - -class LinearActivation(Module): - r"""Fused Linear and activation Module. - """ - __constants__ = ['bias'] - - def __init__(self, in_features, out_features, act='gelu', bias=True): - super(LinearActivation, self).__init__() - self.in_features = in_features - self.out_features = out_features - self.fused_gelu = False - self.fused_tanh = False - if isinstance(act, str) or (sys.version_info[0] == 2 - and isinstance(act, unicode)): - if bias and act == 'gelu': - self.fused_gelu = True - elif bias and act == 'tanh': - self.fused_tanh = True - else: - self.act_fn = ACT2FN[act] - else: - self.act_fn = act - self.weight = Parameter(torch.Tensor(out_features, in_features)) - if bias: - self.bias = Parameter(torch.Tensor(out_features)) - else: - self.register_parameter('bias', None) - self.reset_parameters() - - def reset_parameters(self): - init.kaiming_uniform_(self.weight, a=math.sqrt(5)) - if self.bias is not None: - fan_in, _ = init._calculate_fan_in_and_fan_out(self.weight) - bound = 1 / math.sqrt(fan_in) - init.uniform_(self.bias, -bound, bound) - - def forward(self, input): - if self.fused_gelu: - return bias_gelu(self.bias, F.linear(input, self.weight, None)) - elif self.fused_tanh: - return bias_tanh(self.bias, F.linear(input, self.weight, None)) - else: - return self.act_fn(F.linear(input, self.weight, self.bias)) - - def extra_repr(self): - return 'in_features={}, out_features={}, bias={}'.format( - self.in_features, self.out_features, self.bias is not None) - - -class BertConfig(object): - """Configuration class to store the configuration of a `BertModel`. - """ - def __init__(self, - vocab_size_or_config_json_file, - hidden_size=768, - num_hidden_layers=12, - num_attention_heads=12, - intermediate_size=3072, - hidden_act="gelu", - hidden_dropout_prob=0.1, - attention_probs_dropout_prob=0.1, - max_position_embeddings=512, - type_vocab_size=2, - initializer_range=0.02): - """Constructs BertConfig. - - Args: - vocab_size_or_config_json_file: Vocabulary size of `inputs_ids` in `BertModel`. - hidden_size: Size of the encoder layers and the pooler layer. - num_hidden_layers: Number of hidden layers in the Transformer encoder. - num_attention_heads: Number of attention heads for each attention layer in - the Transformer encoder. - intermediate_size: The size of the "intermediate" (i.e., feed-forward) - layer in the Transformer encoder. - hidden_act: The non-linear activation function (function or string) in the - encoder and pooler. If string, "gelu", "relu" and "swish" are supported. - hidden_dropout_prob: The dropout probabilitiy for all fully connected - layers in the embeddings, encoder, and pooler. - attention_probs_dropout_prob: The dropout ratio for the attention - probabilities. - max_position_embeddings: The maximum sequence length that this model might - ever be used with. Typically set this to something large just in case - (e.g., 512 or 1024 or 2048). - type_vocab_size: The vocabulary size of the `token_type_ids` passed into - `BertModel`. - initializer_range: The sttdev of the truncated_normal_initializer for - initializing all weight matrices. - """ - if isinstance(vocab_size_or_config_json_file, - str) or (sys.version_info[0] == 2 and isinstance( - vocab_size_or_config_json_file, unicode)): - with open(vocab_size_or_config_json_file, "r", - encoding='utf-8') as reader: - json_config = json.loads(reader.read()) - for key, value in json_config.items(): - self.__dict__[key] = value - elif isinstance(vocab_size_or_config_json_file, int): - self.vocab_size = vocab_size_or_config_json_file - self.hidden_size = hidden_size - self.num_hidden_layers = num_hidden_layers - self.num_attention_heads = num_attention_heads - self.hidden_act = hidden_act - self.intermediate_size = intermediate_size - self.hidden_dropout_prob = hidden_dropout_prob - self.attention_probs_dropout_prob = attention_probs_dropout_prob - self.max_position_embeddings = max_position_embeddings - self.type_vocab_size = type_vocab_size - self.initializer_range = initializer_range - else: - raise ValueError( - "First argument must be either a vocabulary size (int)" - "or the path to a pretrained model config file (str)") - - @classmethod - def from_dict(cls, json_object): - """Constructs a `BertConfig` from a Python dictionary of parameters.""" - config = BertConfig(vocab_size_or_config_json_file=-1) - for key, value in json_object.items(): - config.__dict__[key] = value - return config - - @classmethod - def from_json_file(cls, json_file): - """Constructs a `BertConfig` from a json file of parameters.""" - with open(json_file, "r", encoding='utf-8') as reader: - text = reader.read() - return cls.from_dict(json.loads(text)) - - def __repr__(self): - return str(self.to_json_string()) - - def to_dict(self): - """Serializes this instance to a Python dictionary.""" - output = copy.deepcopy(self.__dict__) - return output - - def to_json_string(self): - """Serializes this instance to a JSON string.""" - return json.dumps(self.to_dict(), indent=2, sort_keys=True) + "\n" - - -try: - import apex - #apex.amp.register_half_function(apex.normalization.fused_layer_norm, 'FusedLayerNorm') - import apex.normalization - #apex.amp.register_float_function(apex.normalization.FusedLayerNorm, 'forward') - BertLayerNorm = apex.normalization.FusedLayerNorm -except ImportError: - print( - "Better speed can be achieved with apex installed from https://www.github.com/nvidia/apex." - ) - - class BertLayerNorm(nn.Module): - def __init__(self, hidden_size, eps=1e-12): - """Construct a layernorm module in the TF style (epsilon inside the square root). - """ - super(BertLayerNorm, self).__init__() - self.weight = nn.Parameter(torch.ones(hidden_size)) - self.bias = nn.Parameter(torch.zeros(hidden_size)) - self.variance_epsilon = eps - - def forward(self, x): - pdtype = x.dtype - x = x.float() - u = x.mean(-1, keepdim=True) - s = (x - u).pow(2).mean(-1, keepdim=True) - x = (x - u) / torch.sqrt(s + self.variance_epsilon) - return self.weight * x.to(pdtype) + self.bias - - -class BertEmbeddings(nn.Module): - """Construct the embeddings from word, position and token_type embeddings. - """ - def __init__(self, config): - super(BertEmbeddings, self).__init__() - self.word_embeddings = nn.Embedding(config.vocab_size, - config.hidden_size) - self.position_embeddings = nn.Embedding(config.max_position_embeddings, - config.hidden_size) - self.token_type_embeddings = nn.Embedding(config.type_vocab_size, - config.hidden_size) - - # self.LayerNorm is not snake-cased to stick with TensorFlow model variable name and be able to load - # any TensorFlow checkpoint file - self.LayerNorm = BertLayerNorm(config.hidden_size, eps=1e-12) - self.dropout = nn.Dropout(config.hidden_dropout_prob) - - def forward(self, input_ids, token_type_ids=None): - seq_length = input_ids.size(1) - position_ids = torch.arange(seq_length, - dtype=torch.long, - device=input_ids.device) - position_ids = position_ids.unsqueeze(0).expand_as(input_ids) - if token_type_ids is None: - token_type_ids = torch.zeros_like(input_ids) - - words_embeddings = self.word_embeddings(input_ids) - position_embeddings = self.position_embeddings(position_ids) - token_type_embeddings = self.token_type_embeddings(token_type_ids) - - embeddings = words_embeddings + position_embeddings + token_type_embeddings - embeddings = self.LayerNorm(embeddings) - embeddings = self.dropout(embeddings) - return embeddings - - -class BertSelfAttention(nn.Module): - def __init__(self, config): - super(BertSelfAttention, self).__init__() - if config.hidden_size % config.num_attention_heads != 0: - raise ValueError( - "The hidden size (%d) is not a multiple of the number of attention " - "heads (%d)" % - (config.hidden_size, config.num_attention_heads)) - self.num_attention_heads = config.num_attention_heads - self.attention_head_size = int(config.hidden_size / - config.num_attention_heads) - self.all_head_size = self.num_attention_heads * self.attention_head_size - - self.query = nn.Linear(config.hidden_size, self.all_head_size) - self.key = nn.Linear(config.hidden_size, self.all_head_size) - self.value = nn.Linear(config.hidden_size, self.all_head_size) - - self.dropout = nn.Dropout(config.attention_probs_dropout_prob) - self.softmax = nn.Softmax(dim=-1) - - def transpose_for_scores(self, x): - new_x_shape = x.size()[:-1] + (self.num_attention_heads, - self.attention_head_size) - x = x.view(*new_x_shape) - return x.permute(0, 2, 1, 3) - - def transpose_key_for_scores(self, x): - new_x_shape = x.size()[:-1] + (self.num_attention_heads, - self.attention_head_size) - x = x.view(*new_x_shape) - return x.permute(0, 2, 3, 1) - - def forward(self, hidden_states, attention_mask): - mixed_query_layer = self.query(hidden_states) - mixed_key_layer = self.key(hidden_states) - mixed_value_layer = self.value(hidden_states) - - query_layer = self.transpose_for_scores(mixed_query_layer) - key_layer = self.transpose_key_for_scores(mixed_key_layer) - value_layer = self.transpose_for_scores(mixed_value_layer) - - # Take the dot product between "query" and "key" to get the raw attention scores. - attention_scores = torch.matmul(query_layer, key_layer) - attention_scores = attention_scores / math.sqrt( - self.attention_head_size) - # Apply the attention mask is (precomputed for all layers in BertModel forward() function) - attention_scores = attention_scores + attention_mask - - pdtype = attention_scores.dtype - # Normalize the attention scores to probabilities. - attention_probs = self.softmax(attention_scores) - - # This is actually dropping out entire tokens to attend to, which might - # seem a bit unusual, but is taken from the original Transformer paper. - attention_probs = self.dropout(attention_probs) - - context_layer = torch.matmul(attention_probs, value_layer) - context_layer = context_layer.permute(0, 2, 1, 3).contiguous() - new_context_layer_shape = context_layer.size()[:-2] + ( - self.all_head_size, ) - context_layer = context_layer.view(*new_context_layer_shape) - return context_layer - - -class BertSelfOutput(nn.Module): - def __init__(self, config): - super(BertSelfOutput, self).__init__() - self.dense = nn.Linear(config.hidden_size, config.hidden_size) - self.dense.bert_output_layer = True - self.dropout = nn.Dropout(config.hidden_dropout_prob) - - def forward(self, hidden_states, input_tensor): - hidden_states = self.dense(hidden_states) - hidden_states = self.dropout(hidden_states) - return hidden_states - - -class BertAttention(nn.Module): - def __init__(self, config): - super(BertAttention, self).__init__() - self.self = BertSelfAttention(config) - self.output = BertSelfOutput(config) - - def forward(self, input_tensor, attention_mask): - self_output = self.self(input_tensor, attention_mask) - attention_output = self.output(self_output, input_tensor) - return attention_output - - -class BertIntermediate(nn.Module): - def __init__(self, config): - super(BertIntermediate, self).__init__() - self.dense_act = LinearActivation(config.hidden_size, - config.intermediate_size, - act=config.hidden_act) - - def forward(self, hidden_states): - hidden_states = self.dense_act(hidden_states) - return hidden_states - - -class BertOutput(nn.Module): - def __init__(self, config): - super(BertOutput, self).__init__() - self.dense = nn.Linear(config.intermediate_size, config.hidden_size) - self.dense.bert_output_layer = True - self.dropout = nn.Dropout(config.hidden_dropout_prob) - - def forward(self, hidden_states): - hidden_states = self.dense(hidden_states) - hidden_states = self.dropout(hidden_states) - return hidden_states - - -class BertLayer(nn.Module): - def __init__(self, config): - super(BertLayer, self).__init__() - self.attention = BertAttention(config) - self.PreAttentionLayerNorm = BertLayerNorm(config.hidden_size, - eps=1e-12) - self.PostAttentionLayerNorm = BertLayerNorm(config.hidden_size, - eps=1e-12) - self.intermediate = BertIntermediate(config) - self.output = BertOutput(config) - - def forward(self, hidden_states, attention_mask, action=1, keep_prob=1.0): - if action == 0: - intermediate_input = hidden_states - else: - input_layer_norm = self.PreAttentionLayerNorm(hidden_states) - attention_output = self.attention(input_layer_norm, attention_mask) - attention_output = attention_output * 1 / keep_prob - intermediate_input = hidden_states + attention_output - - if action == 0: - layer_output = intermediate_input - else: - intermediate_layer_norm = self.PostAttentionLayerNorm(intermediate_input) - intermediate_output = self.intermediate(intermediate_layer_norm) - layer_output = self.output(intermediate_output) - layer_output = layer_output * 1 / keep_prob - layer_output = layer_output + intermediate_input - - return layer_output - - -class BertEncoder(nn.Module): - def __init__(self, config, args, sparse_attention_config=None): - super(BertEncoder, self).__init__() - - #Added later to make it similar to GPT-2 - self.FinalLayerNorm = BertLayerNorm(config.hidden_size, eps=1e-12) - - if args.deepspeed_transformer_kernel and args.deepspeed_sparse_attention: - raise NotImplementedError( - f'Currently DeepSpeed Transformer Kernels do not support Sparse Attention. To use Sparse Attention, you need to disable Transformer Kernels!' - ) - - if args.deepspeed_transformer_kernel: - from deepspeed import DeepSpeedTransformerLayer, DeepSpeedTransformerConfig - - ds_config = get_deepspeed_config(args) - cuda_config = DeepSpeedTransformerConfig( - batch_size=ds_config.train_micro_batch_size_per_gpu, - max_seq_length=args.max_seq_length, - hidden_size=config.hidden_size, - intermediate_size=config.intermediate_size, - heads=config.num_attention_heads, - attn_dropout_ratio=config.attention_probs_dropout_prob, - hidden_dropout_ratio=config.hidden_dropout_prob, - num_hidden_layers=config.num_hidden_layers, - initializer_range=config.initializer_range, - local_rank=args.local_rank - if hasattr(args, 'local_rank') else -1, - seed=args.seed, - fp16=ds_config.fp16_enabled, - pre_layer_norm=True, - attn_dropout_checkpoint=args.attention_dropout_checkpoint, - normalize_invertible=args.normalize_invertible, - gelu_checkpoint=args.gelu_checkpoint, - stochastic_mode=args.stochastic_mode) - - self.layer = nn.ModuleList([ - copy.deepcopy(DeepSpeedTransformerLayer(i, cuda_config)) - for i in range(config.num_hidden_layers) - ]) - else: - layer = BertLayer(config) - if sparse_attention_config is not None: - from deepspeed.ops.sparse_attention import BertSparseSelfAttention - - layer.attention.self = BertSparseSelfAttention( - config, sparsity_config=sparse_attention_config) - - self.layer = nn.ModuleList([ - copy.deepcopy(layer) for _ in range(config.num_hidden_layers) - ]) - - # def forward(self, hidden_states, attention_mask, output_all_encoded_layers=True): - # all_encoder_layers = [] - # for layer_module in self.layer: - # hidden_states = layer_module(hidden_states, attention_mask) - # if output_all_encoded_layers: - # all_encoder_layers.append(hidden_states) - # if not output_all_encoded_layers: - # all_encoder_layers.append(hidden_states) - # return all_encoder_layers - def forward(self, - hidden_states, - attention_mask, - output_all_encoded_layers=True, - checkpoint_activations=False, - progressive_layer_drop=False, theta=0.5): - all_encoder_layers = [] - - def custom(start, end): - def custom_forward(*inputs): - layers = self.layer[start:end] - x_ = inputs[0] - for layer in layers: - x_ = layer(x_, inputs[1]) - return x_ - - return custom_forward - - if checkpoint_activations: - l = 0 - num_layers = len(self.layer) - chunk_length = math.ceil(math.sqrt(num_layers)) - while l < num_layers: - hidden_states = checkpoint.checkpoint( - custom(l, l + chunk_length), hidden_states, - attention_mask * 1) - l += chunk_length - # decoder layers - else: - if not progressive_layer_drop: - for i, layer_module in enumerate(self.layer): - hidden_states = layer_module(hidden_states, attention_mask) - - if output_all_encoded_layers: - all_encoder_layers.append(hidden_states) - else: - drop_prob = 1 - theta - step = drop_prob / len(self.layer) - p = 1.0 - - for i,layer_module in enumerate(self.layer): - - action = np.random.choice([1, 0], p=[p, 1-p]) - p = p - step - hidden_states = layer_module(hidden_states, attention_mask, action, p) - if output_all_encoded_layers: - all_encoder_layers.append(hidden_states) - - if not output_all_encoded_layers or checkpoint_activations: - hidden_states = self.FinalLayerNorm(hidden_states) - all_encoder_layers.append(hidden_states) - return all_encoder_layers - - -#class BertEncoder(nn.Module): -# def __init__(self, config): -# super(BertEncoder, self).__init__() -# layer = BertLayer(config) -# self.layer = nn.ModuleList([copy.deepcopy(layer) for _ in range(config.num_hidden_layers)]) -# -# def forward(self, hidden_states, attention_mask, output_all_encoded_layers=True): -# all_encoder_layers = [] -# for layer_module in self.layer: -# hidden_states = layer_module(hidden_states, attention_mask) -# if output_all_encoded_layers: -# all_encoder_layers.append(hidden_states) -# if not output_all_encoded_layers: -# all_encoder_layers.append(hidden_states) -# return all_encoder_layers - - -class BertPooler(nn.Module): - def __init__(self, config): - super(BertPooler, self).__init__() - self.dense_act = LinearActivation(config.hidden_size, - config.hidden_size, - act="tanh") - - def forward(self, hidden_states): - # We "pool" the model by simply taking the hidden state corresponding - # to the first token. - first_token_tensor = hidden_states[:, 0] - pooled_output = self.dense_act(first_token_tensor) - return pooled_output - - -class BertPredictionHeadTransform(nn.Module): - def __init__(self, config): - super(BertPredictionHeadTransform, self).__init__() - self.dense_act = LinearActivation(config.hidden_size, - config.hidden_size, - act=config.hidden_act) - self.LayerNorm = BertLayerNorm(config.hidden_size, eps=1e-12) - - def forward(self, hidden_states): - hidden_states = self.dense_act(hidden_states) - hidden_states = self.LayerNorm(hidden_states) - return hidden_states - - -class BertLMPredictionHead(nn.Module): - def __init__(self, config, bert_model_embedding_weights): - super(BertLMPredictionHead, self).__init__() - self.transform = BertPredictionHeadTransform(config) - - # The output weights are the same as the input embeddings, but there is - # an output-only bias for each token. - self.decoder = nn.Linear(bert_model_embedding_weights.size(1), - bert_model_embedding_weights.size(0), - bias=False) - self.decoder.weight = bert_model_embedding_weights - self.bias = nn.Parameter( - torch.zeros(bert_model_embedding_weights.size(0))) - - def forward(self, hidden_states, masked_token_indexes): - hidden_states = self.transform(hidden_states) - - if masked_token_indexes is not None: - hidden_states = torch.index_select( - hidden_states.view(-1, hidden_states.shape[-1]), 0, - masked_token_indexes) - - torch.cuda.nvtx.range_push( - "decoder input.size() = {}, weight.size() = {}".format( - hidden_states.size(), self.decoder.weight.size())) - hidden_states = self.decoder(hidden_states) + self.bias - torch.cuda.nvtx.range_pop() - return hidden_states - - -class BertOnlyMLMHead(nn.Module): - def __init__(self, config, bert_model_embedding_weights): - super(BertOnlyMLMHead, self).__init__() - self.predictions = BertLMPredictionHead(config, - bert_model_embedding_weights) - - def forward(self, sequence_output): - prediction_scores = self.predictions(sequence_output) - return prediction_scores - - -class BertOnlyNSPHead(nn.Module): - def __init__(self, config): - super(BertOnlyNSPHead, self).__init__() - self.seq_relationship = nn.Linear(config.hidden_size, 2) - - def forward(self, pooled_output): - seq_relationship_score = self.seq_relationship(pooled_output) - return seq_relationship_score - - -class BertPreTrainingHeads(nn.Module): - def __init__(self, config, bert_model_embedding_weights): - super(BertPreTrainingHeads, self).__init__() - self.predictions = BertLMPredictionHead(config, - bert_model_embedding_weights) - self.seq_relationship = nn.Linear(config.hidden_size, 2) - - def forward(self, - sequence_output, - pooled_output, - masked_token_indexes=None): - prediction_scores = self.predictions(sequence_output, - masked_token_indexes) - seq_relationship_score = self.seq_relationship(pooled_output) - return prediction_scores, seq_relationship_score - - -class BertPreTrainedModel(nn.Module): - """ An abstract class to handle weights initialization and - a simple interface for dowloading and loading pretrained models. - """ - def __init__(self, config, *inputs, **kwargs): - super(BertPreTrainedModel, self).__init__() - if not isinstance(config, BertConfig): - raise ValueError( - "Parameter config in `{}(config)` should be an instance of class `BertConfig`. " - "To create a model from a Google pretrained model use " - "`model = {}.from_pretrained(PRETRAINED_MODEL_NAME)`".format( - self.__class__.__name__, self.__class__.__name__)) - self.config = config - - def init_bert_weights(self, module): - """ Initialize the weights. - """ - if isinstance(module, (nn.Linear, nn.Embedding)): - # Slightly different from the TF version which uses truncated_normal for initialization - # cf https://github.com/pytorch/pytorch/pull/5617 - num_layers = self.config.num_hidden_layers - std = self.config.initializer_range - if hasattr(module, 'bert_output_layer'): - # "Accounting for accumulation on the residual path" - #print("Accounting for accumulation on the residual path") - std = self.config.initializer_range / math.sqrt( - 2.0 * num_layers) - module.weight.data.normal_(mean=0.0, std=std) - elif isinstance(module, BertLayerNorm): - module.bias.data.zero_() - module.weight.data.fill_(1.0) - if isinstance(module, nn.Linear) and module.bias is not None: - module.bias.data.zero_() - - @classmethod - def from_pretrained(cls, - pretrained_model_name_or_path, - state_dict=None, - cache_dir=None, - from_tf=False, - *inputs, - **kwargs): - """ - Instantiate a BertPreTrainedModel from a pre-trained model file or a pytorch state dict. - Download and cache the pre-trained model file if needed. - - Params: - pretrained_model_name_or_path: either: - - a str with the name of a pre-trained model to load selected in the list of: - . `bert-base-uncased` - . `bert-large-uncased` - . `bert-base-cased` - . `bert-large-cased` - . `bert-base-multilingual-uncased` - . `bert-base-multilingual-cased` - . `bert-base-chinese` - - a path or url to a pretrained model archive containing: - . `bert_config.json` a configuration file for the model - . `pytorch_model.bin` a PyTorch dump of a BertForPreTraining instance - - a path or url to a pretrained model archive containing: - . `bert_config.json` a configuration file for the model - . `model.chkpt` a TensorFlow checkpoint - from_tf: should we load the weights from a locally saved TensorFlow checkpoint - cache_dir: an optional path to a folder in which the pre-trained models will be cached. - state_dict: an optional state dictionnary (collections.OrderedDict object) to use instead of Google pre-trained models - *inputs, **kwargs: additional input for the specific Bert class - (ex: num_labels for BertForSequenceClassification) - """ - if pretrained_model_name_or_path in PRETRAINED_MODEL_ARCHIVE_MAP: - archive_file = PRETRAINED_MODEL_ARCHIVE_MAP[ - pretrained_model_name_or_path] - else: - archive_file = pretrained_model_name_or_path - # redirect to the cache, if necessary - try: - resolved_archive_file = cached_path(archive_file, - cache_dir=cache_dir) - except EnvironmentError: - logger.error( - "Model name '{}' was not found in model name list ({}). " - "We assumed '{}' was a path or url but couldn't find any file " - "associated to this path or url.".format( - pretrained_model_name_or_path, - ', '.join(PRETRAINED_MODEL_ARCHIVE_MAP.keys()), - archive_file)) - return None - if resolved_archive_file == archive_file: - logger.info("loading archive file {}".format(archive_file)) - else: - logger.info("loading archive file {} from cache at {}".format( - archive_file, resolved_archive_file)) - tempdir = None - if os.path.isdir(resolved_archive_file) or from_tf: - serialization_dir = resolved_archive_file - else: - # Extract archive to temp dir - tempdir = tempfile.mkdtemp() - logger.info("extracting archive file {} to temp dir {}".format( - resolved_archive_file, tempdir)) - with tarfile.open(resolved_archive_file, 'r:gz') as archive: - archive.extractall(tempdir) - serialization_dir = tempdir - # Load config - config_file = os.path.join(serialization_dir, CONFIG_NAME) - config = BertConfig.from_json_file(config_file) - logger.info("Model config {}".format(config)) - # Instantiate model. - model = cls(config, *inputs, **kwargs) - if state_dict is None and not from_tf: - weights_path = os.path.join(serialization_dir, WEIGHTS_NAME) - state_dict = torch.load( - weights_path, - map_location='cpu' if not torch.cuda.is_available() else None) - if tempdir: - # Clean up temp dir - shutil.rmtree(tempdir) - if from_tf: - # Directly load from a TensorFlow checkpoint - weights_path = os.path.join(serialization_dir, TF_WEIGHTS_NAME) - return load_tf_weights_in_bert(model, weights_path) - # Load from a PyTorch state_dict - old_keys = [] - new_keys = [] - for key in state_dict.keys(): - new_key = None - if 'gamma' in key: - new_key = key.replace('gamma', 'weight') - if 'beta' in key: - new_key = key.replace('beta', 'bias') - if new_key: - old_keys.append(key) - new_keys.append(new_key) - for old_key, new_key in zip(old_keys, new_keys): - state_dict[new_key] = state_dict.pop(old_key) - - missing_keys = [] - unexpected_keys = [] - error_msgs = [] - # copy state_dict so _load_from_state_dict can modify it - metadata = getattr(state_dict, '_metadata', None) - state_dict = state_dict.copy() - if metadata is not None: - state_dict._metadata = metadata - - def load(module, prefix=''): - local_metadata = {} if metadata is None else metadata.get( - prefix[:-1], {}) - module._load_from_state_dict(state_dict, prefix, local_metadata, - True, missing_keys, unexpected_keys, - error_msgs) - for name, child in module._modules.items(): - if child is not None: - load(child, prefix + name + '.') - - start_prefix = '' - if not hasattr(model, 'bert') and any( - s.startswith('bert.') for s in state_dict.keys()): - start_prefix = 'bert.' - load(model, prefix=start_prefix) - if len(missing_keys) > 0: - logger.info( - "Weights of {} not initialized from pretrained model: {}". - format(model.__class__.__name__, missing_keys)) - if len(unexpected_keys) > 0: - logger.info( - "Weights from pretrained model not used in {}: {}".format( - model.__class__.__name__, unexpected_keys)) - if len(error_msgs) > 0: - raise RuntimeError( - 'Error(s) in loading state_dict for {}:\n\t{}'.format( - model.__class__.__name__, "\n\t".join(error_msgs))) - return model - - -class BertModel(BertPreTrainedModel): - """BERT model ("Bidirectional Embedding Representations from a Transformer"). - - Params: - config: a BertConfig class instance with the configuration to build a new model - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `output_all_encoded_layers`: boolean which controls the content of the `encoded_layers` output as described below. Default: `True`. - - Outputs: Tuple of (encoded_layers, pooled_output) - `encoded_layers`: controled by `output_all_encoded_layers` argument: - - `output_all_encoded_layers=True`: outputs a list of the full sequences of encoded-hidden-states at the end - of each attention block (i.e. 12 full sequences for BERT-base, 24 for BERT-large), each - encoded-hidden-state is a torch.FloatTensor of size [batch_size, sequence_length, hidden_size], - - `output_all_encoded_layers=False`: outputs only the full sequence of hidden-states corresponding - to the last attention block of shape [batch_size, sequence_length, hidden_size], - `pooled_output`: a torch.FloatTensor of size [batch_size, hidden_size] which is the output of a - classifier pretrained on top of the hidden state associated to the first character of the - input (`CLS`) to train on the Next-Sentence task (see BERT's paper). - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = modeling.BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - model = modeling.BertModel(config=config) - all_encoder_layers, pooled_output = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config, args=None): - super(BertModel, self).__init__(config) - self.embeddings = BertEmbeddings(config) - # set pad_token_id that is used for sparse attention padding - self.pad_token_id = config.pad_token_id if hasattr( - config, 'pad_token_id') and config.pad_token_id is not None else 0 - # set sparse_attention_config if it has been selected - self.sparse_attention_config = get_sparse_attention_config( - args, config.num_attention_heads) - self.sparse_attention_utils = get_sparse_attention_utils(self.sparse_attention_config) - self.encoder = BertEncoder( - config, args, sparse_attention_config=self.sparse_attention_config) - self.pooler = BertPooler(config) - self.apply(self.init_bert_weights) - logger.info("Init BERT pretrain model") - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - output_all_encoded_layers=True, - checkpoint_activations=False, - progressive_layer_drop=False, theta=0.5): - if attention_mask is None: - attention_mask = torch.ones_like(input_ids) - if token_type_ids is None: - token_type_ids = torch.zeros_like(input_ids) - - # We create a 3D attention mask from a 2D tensor mask. - # Sizes are [batch_size, 1, 1, to_seq_length] - # So we can broadcast to [batch_size, num_heads, from_seq_length, to_seq_length] - # this attention mask is more simple than the triangular masking of causal attention - # used in OpenAI GPT, we just need to prepare the broadcast dimension here. - extended_attention_mask = attention_mask.unsqueeze(1).unsqueeze(2) - - # Since attention_mask is 1.0 for positions we want to attend and 0.0 for - # masked positions, this operation will create a tensor which is 0.0 for - # positions we want to attend and -10000.0 for masked positions. - # Since we are adding it to the raw scores before the softmax, this is - # effectively the same as removing these entirely. - extended_attention_mask = extended_attention_mask.to( - dtype=next(self.parameters()).dtype) # fp16 compatibility - extended_attention_mask = (1.0 - extended_attention_mask) * -10000.0 - - # If BertEncoder uses sparse attention, it needs to be padded based on the sparse attention block size - if self.sparse_attention_config is not None: - pad_len, input_ids, attention_mask, token_type_ids, position_ids, inputs_embeds = self.sparse_attention_utils.pad_to_block_size( - block_size=self.sparse_attention_config.block, - input_ids=input_ids, - attention_mask=extended_attention_mask, - token_type_ids=token_type_ids, - position_ids=None, - inputs_embeds=None, - pad_token_id=self.pad_token_id, - model_mbeddings=self.embeddings) - - embedding_output = self.embeddings(input_ids, token_type_ids) - encoded_layers = self.encoder( - embedding_output, - extended_attention_mask, - output_all_encoded_layers=output_all_encoded_layers, - checkpoint_activations=checkpoint_activations, - progressive_layer_drop=progressive_layer_drop, theta=theta) - sequence_output = encoded_layers[-1] - pooled_output = self.pooler(sequence_output) - - # If BertEncoder uses sparse attention, and input_ids were padded, sequence output needs to be unpadded to original length - if self.sparse_attention_config is not None and pad_len > 0: - encoded_layers[-1] = self.sparse_attention_utils.unpad_sequence_output( - pad_len, encoded_layers[-1]) - - if not output_all_encoded_layers: - encoded_layers = encoded_layers[-1] - return encoded_layers, pooled_output - - -class BertForPreTrainingPreLN(BertPreTrainedModel): - """BERT model with pre-training heads. - This module comprises the BERT model followed by the two pre-training heads: - - the masked language modeling head, and - - the next sentence classification head. - - Params: - config: a BertConfig class instance with the configuration to build a new model. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `masked_lm_labels`: optional masked language modeling labels: torch.LongTensor of shape [batch_size, sequence_length] - with indices selected in [-1, 0, ..., vocab_size]. All labels set to -1 are ignored (masked), the loss - is only computed for the labels set in [0, ..., vocab_size] - `next_sentence_label`: optional next sentence classification loss: torch.LongTensor of shape [batch_size] - with indices selected in [0, 1]. - 0 => next sentence is the continuation, 1 => next sentence is a random sentence. - - Outputs: - if `masked_lm_labels` and `next_sentence_label` are not `None`: - Outputs the total_loss which is the sum of the masked language modeling loss and the next - sentence classification loss. - if `masked_lm_labels` or `next_sentence_label` is `None`: - Outputs a tuple comprising - - the masked language modeling logits of shape [batch_size, sequence_length, vocab_size], and - - the next sentence classification logits of shape [batch_size, 2]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - model = BertForPreTraining(config) - masked_lm_logits_scores, seq_relationship_logits = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config, args): - super(BertForPreTrainingPreLN, self).__init__(config) - self.bert = BertModel(config, args) - self.cls = BertPreTrainingHeads( - config, self.bert.embeddings.word_embeddings.weight) - self.apply(self.init_bert_weights) - self.args = args - - def forward(self, batch, **kwargs): - progressive_layer_drop = kwargs['progressive_layer_drop'] - theta = kwargs['theta'] - log = kwargs['log'] - - input_ids = batch[1] - token_type_ids = batch[3] - attention_mask = batch[2] - masked_lm_labels = batch[5] - next_sentence_label = batch[4] - checkpoint_activations = False - - sequence_output, pooled_output = self.bert( - input_ids, - token_type_ids, - attention_mask, - output_all_encoded_layers=False, - checkpoint_activations=checkpoint_activations, - progressive_layer_drop=progressive_layer_drop, theta=theta) - - if masked_lm_labels is not None and next_sentence_label is not None: - # filter out all masked labels. - masked_token_indexes = torch.nonzero( - (masked_lm_labels + 1).view(-1)).view(-1) - prediction_scores, seq_relationship_score = self.cls( - sequence_output, pooled_output, masked_token_indexes) - target = torch.index_select(masked_lm_labels.view(-1), 0, - masked_token_indexes) - - loss_fct = CrossEntropyLoss(ignore_index=-1) - masked_lm_loss = loss_fct( - prediction_scores.view(-1, self.config.vocab_size), target) - next_sentence_loss = loss_fct(seq_relationship_score.view(-1, 2), - next_sentence_label.view(-1)) - total_loss = masked_lm_loss + next_sentence_loss - return total_loss - else: - prediction_scores, seq_relationship_score = self.cls( - sequence_output, pooled_output) - return prediction_scores, seq_relationship_score - - -class BertForMaskedLM(BertPreTrainedModel): - """BERT model with the masked language modeling head. - This module comprises the BERT model followed by the masked language modeling head. - - Params: - config: a BertConfig class instance with the configuration to build a new model. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `masked_lm_labels`: masked language modeling labels: torch.LongTensor of shape [batch_size, sequence_length] - with indices selected in [-1, 0, ..., vocab_size]. All labels set to -1 are ignored (masked), the loss - is only computed for the labels set in [0, ..., vocab_size] - - Outputs: - if `masked_lm_labels` is not `None`: - Outputs the masked language modeling loss. - if `masked_lm_labels` is `None`: - Outputs the masked language modeling logits of shape [batch_size, sequence_length, vocab_size]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - model = BertForMaskedLM(config) - masked_lm_logits_scores = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config): - super(BertForMaskedLM, self).__init__(config) - self.bert = BertModel(config) - self.cls = BertOnlyMLMHead(config, - self.bert.embeddings.word_embeddings.weight) - self.apply(self.init_bert_weights) - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - masked_lm_labels=None, - checkpoint_activations=False): - sequence_output, _ = self.bert(input_ids, - token_type_ids, - attention_mask, - output_all_encoded_layers=False) - prediction_scores = self.cls(sequence_output) - - if masked_lm_labels is not None: - loss_fct = CrossEntropyLoss(ignore_index=-1) - masked_lm_loss = loss_fct( - prediction_scores.view(-1, self.config.vocab_size), - masked_lm_labels.view(-1)) - return masked_lm_loss - else: - return prediction_scores - - -class BertForNextSentencePrediction(BertPreTrainedModel): - """BERT model with next sentence prediction head. - This module comprises the BERT model followed by the next sentence classification head. - - Params: - config: a BertConfig class instance with the configuration to build a new model. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `next_sentence_label`: next sentence classification loss: torch.LongTensor of shape [batch_size] - with indices selected in [0, 1]. - 0 => next sentence is the continuation, 1 => next sentence is a random sentence. - - Outputs: - if `next_sentence_label` is not `None`: - Outputs the total_loss which is the sum of the masked language modeling loss and the next - sentence classification loss. - if `next_sentence_label` is `None`: - Outputs the next sentence classification logits of shape [batch_size, 2]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - model = BertForNextSentencePrediction(config) - seq_relationship_logits = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config): - super(BertForNextSentencePrediction, self).__init__(config) - self.bert = BertModel(config) - self.cls = BertOnlyNSPHead(config) - self.apply(self.init_bert_weights) - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - next_sentence_label=None, - checkpoint_activations=False): - _, pooled_output = self.bert(input_ids, - token_type_ids, - attention_mask, - output_all_encoded_layers=False) - seq_relationship_score = self.cls(pooled_output) - - if next_sentence_label is not None: - loss_fct = CrossEntropyLoss(ignore_index=-1) - next_sentence_loss = loss_fct(seq_relationship_score.view(-1, 2), - next_sentence_label.view(-1)) - return next_sentence_loss - else: - return seq_relationship_score - - -class BertForSequenceClassification(BertPreTrainedModel): - """BERT model for classification. - This module is composed of the BERT model with a linear layer on top of - the pooled output. - - Params: - `config`: a BertConfig class instance with the configuration to build a new model. - `num_labels`: the number of classes for the classifier. Default = 2. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `labels`: labels for the classification output: torch.LongTensor of shape [batch_size] - with indices selected in [0, ..., num_labels]. - - Outputs: - if `labels` is not `None`: - Outputs the CrossEntropy classification loss of the output with the labels. - if `labels` is `None`: - Outputs the classification logits of shape [batch_size, num_labels]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - num_labels = 2 - - model = BertForSequenceClassification(config, num_labels) - logits = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config, num_labels): - super(BertForSequenceClassification, self).__init__(config) - self.num_labels = num_labels - self.bert = BertModel(config) - self.dropout = nn.Dropout(config.hidden_dropout_prob) - self.classifier = nn.Linear(config.hidden_size, num_labels) - self.apply(self.init_bert_weights) - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - labels=None, - checkpoint_activations=False): - _, pooled_output = self.bert(input_ids, - token_type_ids, - attention_mask, - output_all_encoded_layers=False) - pooled_output = self.dropout(pooled_output) - logits = self.classifier(pooled_output) - - if labels is not None: - loss_fct = CrossEntropyLoss() - loss = loss_fct(logits.view(-1, self.num_labels), labels.view(-1)) - return loss - else: - return logits - - -class BertForMultipleChoice(BertPreTrainedModel): - """BERT model for multiple choice tasks. - This module is composed of the BERT model with a linear layer on top of - the pooled output. - - Params: - `config`: a BertConfig class instance with the configuration to build a new model. - `num_choices`: the number of classes for the classifier. Default = 2. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, num_choices, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, num_choices, sequence_length] - with the token types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` - and type 1 corresponds to a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, num_choices, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `labels`: labels for the classification output: torch.LongTensor of shape [batch_size] - with indices selected in [0, ..., num_choices]. - - Outputs: - if `labels` is not `None`: - Outputs the CrossEntropy classification loss of the output with the labels. - if `labels` is `None`: - Outputs the classification logits of shape [batch_size, num_labels]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[[31, 51, 99], [15, 5, 0]], [[12, 16, 42], [14, 28, 57]]]) - input_mask = torch.LongTensor([[[1, 1, 1], [1, 1, 0]],[[1,1,0], [1, 0, 0]]]) - token_type_ids = torch.LongTensor([[[0, 0, 1], [0, 1, 0]],[[0, 1, 1], [0, 0, 1]]]) - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - num_choices = 2 - - model = BertForMultipleChoice(config, num_choices) - logits = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config, num_choices): - super(BertForMultipleChoice, self).__init__(config) - self.num_choices = num_choices - self.bert = BertModel(config) - self.dropout = nn.Dropout(config.hidden_dropout_prob) - self.classifier = nn.Linear(config.hidden_size, 1) - self.apply(self.init_bert_weights) - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - labels=None, - checkpoint_activations=False): - flat_input_ids = input_ids.view(-1, input_ids.size(-1)) - flat_token_type_ids = token_type_ids.view(-1, token_type_ids.size(-1)) - flat_attention_mask = attention_mask.view(-1, attention_mask.size(-1)) - _, pooled_output = self.bert(flat_input_ids, - flat_token_type_ids, - flat_attention_mask, - output_all_encoded_layers=False) - pooled_output = self.dropout(pooled_output) - logits = self.classifier(pooled_output) - reshaped_logits = logits.view(-1, self.num_choices) - - if labels is not None: - loss_fct = CrossEntropyLoss() - loss = loss_fct(reshaped_logits, labels) - return loss - else: - return reshaped_logits - - -class BertForTokenClassification(BertPreTrainedModel): - """BERT model for token-level classification. - This module is composed of the BERT model with a linear layer on top of - the full hidden state of the last layer. - - Params: - `config`: a BertConfig class instance with the configuration to build a new model. - `num_labels`: the number of classes for the classifier. Default = 2. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `labels`: labels for the classification output: torch.LongTensor of shape [batch_size, sequence_length] - with indices selected in [0, ..., num_labels]. - - Outputs: - if `labels` is not `None`: - Outputs the CrossEntropy classification loss of the output with the labels. - if `labels` is `None`: - Outputs the classification logits of shape [batch_size, sequence_length, num_labels]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - num_labels = 2 - - model = BertForTokenClassification(config, num_labels) - logits = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config, num_labels): - super(BertForTokenClassification, self).__init__(config) - self.num_labels = num_labels - self.bert = BertModel(config) - self.dropout = nn.Dropout(config.hidden_dropout_prob) - self.classifier = nn.Linear(config.hidden_size, num_labels) - self.apply(self.init_bert_weights) - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - labels=None, - checkpoint_activations=False): - sequence_output, _ = self.bert(input_ids, - token_type_ids, - attention_mask, - output_all_encoded_layers=False) - sequence_output = self.dropout(sequence_output) - logits = self.classifier(sequence_output) - - if labels is not None: - loss_fct = CrossEntropyLoss() - # Only keep active parts of the loss - if attention_mask is not None: - active_loss = attention_mask.view(-1) == 1 - active_logits = logits.view(-1, self.num_labels)[active_loss] - active_labels = labels.view(-1)[active_loss] - loss = loss_fct(active_logits, active_labels) - else: - loss = loss_fct(logits.view(-1, self.num_labels), - labels.view(-1)) - return loss - else: - return logits - - -class BertForQuestionAnswering(BertPreTrainedModel): - """BERT model for Question Answering (span extraction). - This module is composed of the BERT model with a linear layer on top of - the sequence output that computes start_logits and end_logits - - Params: - `config`: a BertConfig class instance with the configuration to build a new model. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `start_positions`: position of the first token for the labeled span: torch.LongTensor of shape [batch_size]. - Positions are clamped to the length of the sequence and position outside of the sequence are not taken - into account for computing the loss. - `end_positions`: position of the last token for the labeled span: torch.LongTensor of shape [batch_size]. - Positions are clamped to the length of the sequence and position outside of the sequence are not taken - into account for computing the loss. - - Outputs: - if `start_positions` and `end_positions` are not `None`: - Outputs the total_loss which is the sum of the CrossEntropy loss for the start and end token positions. - if `start_positions` or `end_positions` is `None`: - Outputs a tuple of start_logits, end_logits which are the logits respectively for the start and end - position tokens of shape [batch_size, sequence_length]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - model = BertForQuestionAnswering(config) - start_logits, end_logits = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config): - super(BertForQuestionAnswering, self).__init__(config) - self.bert = BertModel(config) - # TODO check with Google if it's normal there is no dropout on the token classifier of SQuAD in the TF version - # self.dropout = nn.Dropout(config.hidden_dropout_prob) - self.qa_outputs = nn.Linear(config.hidden_size, 2) - self.apply(self.init_bert_weights) - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - start_positions=None, - end_positions=None, - checkpoint_activations=False): - sequence_output, _ = self.bert(input_ids, - token_type_ids, - attention_mask, - output_all_encoded_layers=False) - logits = self.qa_outputs(sequence_output) - start_logits, end_logits = logits.split(1, dim=-1) - start_logits = start_logits.squeeze(-1) - end_logits = end_logits.squeeze(-1) - - if start_positions is not None and end_positions is not None: - # If we are on multi-GPU, split add a dimension - if len(start_positions.size()) > 1: - start_positions = start_positions.squeeze(-1) - if len(end_positions.size()) > 1: - end_positions = end_positions.squeeze(-1) - # sometimes the start/end positions are outside our model inputs, we ignore these terms - ignored_index = start_logits.size(1) - start_positions.clamp_(0, ignored_index) - end_positions.clamp_(0, ignored_index) - - loss_fct = CrossEntropyLoss(ignore_index=ignored_index) - start_loss = loss_fct(start_logits, start_positions) - end_loss = loss_fct(end_logits, end_positions) - total_loss = (start_loss + end_loss) / 2 - return total_loss - else: - return start_logits, end_logits diff --git a/training/MoQ/README.md b/training/MoQ/README.md deleted file mode 100644 index 7bf1ce992..000000000 --- a/training/MoQ/README.md +++ /dev/null @@ -1,5 +0,0 @@ -# Not maintained / deprecated - -> __Warning__ -> This folder/feature has been deprecated. Feel free to test and submit an issue if you run into errors. - diff --git a/training/MoQ/huggingface-transformers/examples/research_projects/lxmert/requirements.txt b/training/MoQ/huggingface-transformers/examples/research_projects/lxmert/requirements.txt deleted file mode 100644 index 69bc6ba07..000000000 --- a/training/MoQ/huggingface-transformers/examples/research_projects/lxmert/requirements.txt +++ /dev/null @@ -1,98 +0,0 @@ -appdirs==1.4.3 -argon2-cffi==20.1.0 -async-generator==1.10 -attrs==20.2.0 -backcall==0.2.0 -CacheControl==0.12.6 -certifi==2020.6.20 -cffi==1.14.2 -chardet==3.0.4 -click==7.1.2 -colorama==0.4.3 -contextlib2==0.6.0 -cycler==0.10.0 -datasets==1.0.0 -decorator==4.4.2 -defusedxml==0.6.0 -dill==0.3.2 -distlib==0.3.0 -distro==1.4.0 -entrypoints==0.3 -filelock==3.0.12 -future==0.18.2 -html5lib==1.0.1 -idna==2.8 -ipaddr==2.2.0 -ipykernel==5.3.4 -ipython -ipython-genutils==0.2.0 -ipywidgets==7.5.1 -jedi==0.17.2 -Jinja2==2.11.2 -joblib==0.16.0 -jsonschema==3.2.0 -jupyter==1.0.0 -jupyter-client==6.1.7 -jupyter-console==6.2.0 -jupyter-core==4.6.3 -jupyterlab-pygments==0.1.1 -kiwisolver==1.2.0 -lockfile==0.12.2 -MarkupSafe==1.1.1 -matplotlib==3.3.1 -mistune==0.8.4 -msgpack==0.6.2 -nbclient==0.5.0 -nbconvert==6.0.1 -nbformat==5.0.7 -nest-asyncio==1.4.0 -notebook==6.1.5 -numpy==1.19.2 -opencv-python==4.10.0.84 -packaging==20.3 -pandas==1.1.2 -pandocfilters==1.4.2 -parso==0.7.1 -pep517==0.8.2 -pexpect==4.8.0 -pickleshare==0.7.5 -Pillow==7.2.0 -progress==1.5 -prometheus-client==0.8.0 -prompt-toolkit==3.0.7 -ptyprocess==0.6.0 -pyaml==20.4.0 -pyarrow==1.0.1 -pycparser==2.20 -Pygments==2.6.1 -pyparsing==2.4.6 -pyrsistent==0.16.0 -python-dateutil==2.8.1 -pytoml==0.1.21 -pytz==2020.1 -PyYAML==5.3.1 -pyzmq==19.0.2 -qtconsole==4.7.7 -QtPy==1.9.0 -regex==2020.7.14 -requests==2.22.0 -retrying==1.3.3 -sacremoses==0.0.43 -Send2Trash==1.5.0 -sentencepiece==0.1.91 -six==1.14.0 -terminado==0.8.3 -testpath==0.4.4 -tokenizers==0.8.1rc2 -torch==1.13.1 -torchvision==0.7.0 -tornado==6.0.4 -tqdm==4.48.2 -traitlets -git+https://github.com/huggingface/transformers.git -urllib3==1.25.8 -wcwidth==0.2.5 -webencodings==0.5.1 -wget==3.2 -widgetsnbextension==3.5.1 -xxhash==2.0.0 diff --git a/training/MoQ/requirements.txt b/training/MoQ/requirements.txt deleted file mode 100644 index 0f5c38bd4..000000000 --- a/training/MoQ/requirements.txt +++ /dev/null @@ -1,3 +0,0 @@ -datasets >= 1.1.3 -sentencepiece != 0.1.92 -protobuf diff --git a/training/MoQ/run.sh b/training/MoQ/run.sh deleted file mode 100644 index bcae746d7..000000000 --- a/training/MoQ/run.sh +++ /dev/null @@ -1,42 +0,0 @@ -OOO=output -MASTER_PORT=12345 -GPU=0 - -for TSK in qnli #stsb mrpc cola wnli sst2 rte qnli qqp mnli -do - -if [ $TSK == wnli ] || [ $TSK == mrpc ] -then - EPOCH_NUM=5 -else - EPOCH_NUM=3 -fi - -if [ $TSK == qqp ] || [ $TSK == mnli ] -then - TEST_JSON=test_long.json -else - TEST_JSON=test.json -fi - -PORT=$((MASTER_PORT+GPU)) - -rm -rvf ./$OOO/${TSK} - -CUDA_VISIBLE_DEVICES=$GPU python -m torch.distributed.launch \ - --master_port $PORT \ - --nproc_per_node 1 run_glue.py \ - --model_name_or_path bert-base-cased \ - --task_name $TSK \ - --do_train \ - --do_eval \ - --max_seq_length 128 \ - --per_device_train_batch_size 32 \ - --learning_rate 2e-5 \ - --num_train_epochs $EPOCH_NUM \ - --output_dir ./$OOO/$TSK/ \ - --fp16 \ - --warmup_steps 2 \ - --deepspeed test.json - -done diff --git a/training/MoQ/run_glue.py b/training/MoQ/run_glue.py deleted file mode 100644 index 4d18e5ed6..000000000 --- a/training/MoQ/run_glue.py +++ /dev/null @@ -1,561 +0,0 @@ -#!/usr/bin/env python -# coding=utf-8 -# Copyright 2020 The HuggingFace Inc. team. All rights reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -""" Finetuning the library models for sequence classification on GLUE.""" -# You can also adapt this script on your own text classification task. Pointers for this are left as comments. - -import logging -import os -import random -import sys -from dataclasses import dataclass, field -from typing import Optional - -import numpy as np -from datasets import load_dataset, load_metric - -import transformers -from transformers import ( - AutoConfig, - AutoModelForSequenceClassification, - AutoTokenizer, - DataCollatorWithPadding, - EvalPrediction, - HfArgumentParser, - PretrainedConfig, - Trainer, - TrainingArguments, - default_data_collator, - set_seed, -) -from transformers.trainer_utils import get_last_checkpoint, is_main_process - -ds_inference = False - -task_to_keys = { - "cola": ("sentence", - None), - "mnli": ("premise", - "hypothesis"), - "mrpc": ("sentence1", - "sentence2"), - "qnli": ("question", - "sentence"), - "qqp": ("question1", - "question2"), - "rte": ("sentence1", - "sentence2"), - "sst2": ("sentence", - None), - "stsb": ("sentence1", - "sentence2"), - "wnli": ("sentence1", - "sentence2"), -} - -logger = logging.getLogger(__name__) - - -@dataclass -class DataTrainingArguments: - """ - Arguments pertaining to what data we are going to input our model for training and eval. - - Using `HfArgumentParser` we can turn this class - into argparse arguments to be able to specify them on - the command line. - """ - - task_name: Optional[str] = field( - default=None, - metadata={ - "help": "The name of the task to train on: " + ", ".join(task_to_keys.keys()) - }, - ) - max_seq_length: int = field( - default=128, - metadata={ - "help": - "The maximum total input sequence length after tokenization. Sequences longer " - "than this will be truncated, sequences shorter will be padded." - }, - ) - overwrite_cache: bool = field( - default=False, - metadata={"help": "Overwrite the cached preprocessed datasets or not."}) - pad_to_max_length: bool = field( - default=True, - metadata={ - "help": - "Whether to pad all samples to `max_seq_length`. " - "If False, will pad the samples dynamically when batching to the maximum length in the batch." - }, - ) - train_file: Optional[str] = field( - default=None, - metadata={"help": "A csv or a json file containing the training data."}) - validation_file: Optional[str] = field( - default=None, - metadata={"help": "A csv or a json file containing the validation data."}) - test_file: Optional[str] = field( - default=None, - metadata={"help": "A csv or a json file containing the test data."}) - - def __post_init__(self): - if self.task_name is not None: - self.task_name = self.task_name.lower() - if self.task_name not in task_to_keys.keys(): - raise ValueError("Unknown task, you should pick one in " + - ",".join(task_to_keys.keys())) - elif self.train_file is None or self.validation_file is None: - raise ValueError("Need either a GLUE task or a training/validation file.") - else: - train_extension = self.train_file.split(".")[-1] - assert train_extension in ["csv", "json"], "`train_file` should be a csv or a json file." - validation_extension = self.validation_file.split(".")[-1] - assert ( - validation_extension == train_extension - ), "`validation_file` should have the same extension (csv or json) as `train_file`." - - -@dataclass -class ModelArguments: - """ - Arguments pertaining to which model/config/tokenizer we are going to fine-tune from. - """ - - model_name_or_path: str = field(metadata={ - "help": - "Path to pretrained model or model identifier from huggingface.co/models" - }) - config_name: Optional[str] = field( - default=None, - metadata={ - "help": "Pretrained config name or path if not the same as model_name" - }) - tokenizer_name: Optional[str] = field( - default=None, - metadata={ - "help": "Pretrained tokenizer name or path if not the same as model_name" - }) - cache_dir: Optional[str] = field( - default=None, - metadata={ - "help": - "Where do you want to store the pretrained models downloaded from huggingface.co" - }, - ) - use_fast_tokenizer: bool = field( - default=True, - metadata={ - "help": - "Whether to use one of the fast tokenizer (backed by the tokenizers library) or not." - }, - ) - model_revision: str = field( - default="main", - metadata={ - "help": - "The specific model version to use (can be a branch name, tag name or commit id)." - }, - ) - use_auth_token: bool = field( - default=False, - metadata={ - "help": - "Will use the token generated when running `transformers-cli login` (necessary to use this script " - "with private models)." - }, - ) - - -def main(): - # See all possible arguments in src/transformers/training_args.py - # or by passing the --help flag to this script. - # We now keep distinct sets of args, for a cleaner separation of concerns. - - parser = HfArgumentParser((ModelArguments, DataTrainingArguments, TrainingArguments)) - if len(sys.argv) == 2 and sys.argv[1].endswith(".json"): - # If we pass only one argument to the script and it's the path to a json file, - # let's parse it to get our arguments. - model_args, data_args, training_args = parser.parse_json_file(json_file=os.path.abspath(sys.argv[1])) - else: - model_args, data_args, training_args = parser.parse_args_into_dataclasses() - - # Detecting last checkpoint. - last_checkpoint = None - if os.path.isdir( - training_args.output_dir - ) and training_args.do_train and not training_args.overwrite_output_dir: - last_checkpoint = get_last_checkpoint(training_args.output_dir) - if last_checkpoint is None and len(os.listdir(training_args.output_dir)) > 0: - raise ValueError( - f"Output directory ({training_args.output_dir}) already exists and is not empty. " - "Use --overwrite_output_dir to overcome.") - elif last_checkpoint is not None: - logger.info( - f"Checkpoint detected, resuming training at {last_checkpoint}. To avoid this behavior, change " - "the `--output_dir` or add `--overwrite_output_dir` to train from scratch." - ) - - # Setup logging - logging.basicConfig( - format="%(asctime)s - %(levelname)s - %(name)s - %(message)s", - datefmt="%m/%d/%Y %H:%M:%S", - handlers=[logging.StreamHandler(sys.stdout)], - ) - logger.setLevel( - logging.INFO if is_main_process(training_args.local_rank) else logging.WARN) - - # Log on each process the small summary: - logger.warning( - f"Process rank: {training_args.local_rank}, device: {training_args.device}, n_gpu: {training_args.n_gpu}" - + - f"distributed training: {bool(training_args.local_rank != -1)}, 16-bits training: {training_args.fp16}" - ) - # Set the verbosity to info of the Transformers logger (on main process only): - if is_main_process(training_args.local_rank): - transformers.utils.logging.set_verbosity_info() - transformers.utils.logging.enable_default_handler() - transformers.utils.logging.enable_explicit_format() - logger.info(f"Training/evaluation parameters {training_args}") - - # Set seed before initializing model. - set_seed(training_args.seed) - - # Get the datasets: you can either provide your own CSV/JSON training and evaluation files (see below) - # or specify a GLUE benchmark task (the dataset will be downloaded automatically from the datasets Hub). - # - # For CSV/JSON files, this script will use as labels the column called 'label' and as pair of sentences the - # sentences in columns called 'sentence1' and 'sentence2' if such column exists or the first two columns not named - # label if at least two columns are provided. - # - # If the CSVs/JSONs contain only one non-label column, the script does single sentence classification on this - # single column. You can easily tweak this behavior (see below) - # - # In distributed training, the load_dataset function guarantee that only one local process can concurrently - # download the dataset. - if data_args.task_name is not None: - # Downloading and loading a dataset from the hub. - datasets = load_dataset("glue", data_args.task_name) - else: - # Loading a dataset from your local files. - # CSV/JSON training and evaluation files are needed. - data_files = { - "train": data_args.train_file, - "validation": data_args.validation_file - } - - # Get the test dataset: you can provide your own CSV/JSON test file (see below) - # when you use `do_predict` without specifying a GLUE benchmark task. - if training_args.do_predict: - if data_args.test_file is not None: - train_extension = data_args.train_file.split(".")[-1] - test_extension = data_args.test_file.split(".")[-1] - assert ( - test_extension == train_extension - ), "`test_file` should have the same extension (csv or json) as `train_file`." - data_files["test"] = data_args.test_file - else: - raise ValueError( - "Need either a GLUE task or a test file for `do_predict`.") - - for key in data_files.keys(): - logger.info(f"load a local file for {key}: {data_files[key]}") - - if data_args.train_file.endswith(".csv"): - # Loading a dataset from local csv files - datasets = load_dataset("csv", data_files=data_files) - else: - # Loading a dataset from local json files - datasets = load_dataset("json", data_files=data_files) - # See more about loading any type of standard or custom dataset at - # https://huggingface.co/docs/datasets/loading_datasets.html. - - # Labels - if data_args.task_name is not None: - is_regression = data_args.task_name == "stsb" - if not is_regression: - label_list = datasets["train"].features["label"].names - num_labels = len(label_list) - else: - num_labels = 1 - else: - # Trying to have good defaults here, don't hesitate to tweak to your needs. - is_regression = datasets["train"].features["label"].dtype in [ - "float32", - "float64" - ] - if is_regression: - num_labels = 1 - else: - # A useful fast method: - # https://huggingface.co/docs/datasets/package_reference/main_classes.html#datasets.Dataset.unique - label_list = datasets["train"].unique("label") - label_list.sort() # Let's sort it for determinism - num_labels = len(label_list) - - # Load pretrained model and tokenizer - # - # In distributed training, the .from_pretrained methods guarantee that only one local process can concurrently - # download model & vocab. - config = AutoConfig.from_pretrained( - model_args.config_name - if model_args.config_name else model_args.model_name_or_path, - num_labels=num_labels, - finetuning_task=data_args.task_name, - cache_dir=model_args.cache_dir, - revision=model_args.model_revision, - use_auth_token=True if model_args.use_auth_token else None, - ) - tokenizer = AutoTokenizer.from_pretrained( - model_args.tokenizer_name - if model_args.tokenizer_name else model_args.model_name_or_path, - cache_dir=model_args.cache_dir, - use_fast=model_args.use_fast_tokenizer, - revision=model_args.model_revision, - use_auth_token=True if model_args.use_auth_token else None, - ) - model = AutoModelForSequenceClassification.from_pretrained( - model_args.model_name_or_path, - from_tf=bool(".ckpt" in model_args.model_name_or_path), - config=config, - cache_dir=model_args.cache_dir, - revision=model_args.model_revision, - use_auth_token=True if model_args.use_auth_token else None, - ) - - if not training_args.do_train: - import torch - # loading the model from the MoQ-trained checkpoint - sd = torch.load('output/qnli/pytorch_model.bin') - model.load_state_dict(sd) - - import deepspeed - import deepspeed.module_inject as module_inject - deepspeed.init_inference(model, - mp_size=1, - dtype=torch.int8, - replace_method='auto', - quantization_setting=8) - - # Preprocessing the datasets - - if data_args.task_name is not None: - sentence1_key, sentence2_key = task_to_keys[data_args.task_name] - else: - # Again, we try to have some nice defaults but don't hesitate to tweak to your use case. - non_label_column_names = [ - name for name in datasets["train"].column_names if name != "label" - ] - if "sentence1" in non_label_column_names and "sentence2" in non_label_column_names: - sentence1_key, sentence2_key = "sentence1", "sentence2" - else: - if len(non_label_column_names) >= 2: - sentence1_key, sentence2_key = non_label_column_names[:2] - else: - sentence1_key, sentence2_key = non_label_column_names[0], None - - # Padding strategy - if data_args.pad_to_max_length: - padding = "max_length" - else: - # We will pad later, dynamically at batch creation, to the max sequence length in each batch - padding = False - - # Some models have set the order of the labels to use, so let's make sure we do use it. - label_to_id = None - if (model.config.label2id != PretrainedConfig(num_labels=num_labels).label2id - and data_args.task_name is not None and not is_regression): - # Some have all caps in their config, some don't. - label_name_to_id = {k.lower(): v for k, v in model.config.label2id.items()} - if list(sorted(label_name_to_id.keys())) == list(sorted(label_list)): - label_to_id = {i: label_name_to_id[label_list[i]] for i in range(num_labels)} - else: - logger.warn( - "Your model seems to have been trained with labels, but they don't match the dataset: ", - f"model labels: {list(sorted(label_name_to_id.keys()))}, dataset labels: {list(sorted(label_list))}." - "\nIgnoring the model labels as a result.", - ) - elif data_args.task_name is None and not is_regression: - label_to_id = {v: i for i, v in enumerate(label_list)} - - if data_args.max_seq_length > tokenizer.model_max_length: - logger.warn( - f"The max_seq_length passed ({data_args.max_seq_length}) is larger than the maximum length for the" - f"model ({tokenizer.model_max_length}). Using max_seq_length={tokenizer.model_max_length}." - ) - max_seq_length = min(data_args.max_seq_length, tokenizer.model_max_length) - - def preprocess_function(examples): - # Tokenize the texts - args = ((examples[sentence1_key], - ) if sentence2_key is None else (examples[sentence1_key], - examples[sentence2_key])) - result = tokenizer(*args, - padding=padding, - max_length=max_seq_length, - truncation=True) - - # Map labels to IDs (not necessary for GLUE tasks) - if label_to_id is not None and "label" in examples: - result["label"] = [label_to_id[l] for l in examples["label"]] - return result - - datasets = datasets.map(preprocess_function, - batched=True, - load_from_cache_file=not data_args.overwrite_cache) - - train_dataset = datasets["train"] - eval_dataset = datasets["validation_matched" if data_args.task_name == - "mnli" else "validation"] - if data_args.task_name is not None or data_args.test_file is not None: - test_dataset = datasets["test_matched" if data_args.task_name == - "mnli" else "test"] - - # Log a few random samples from the training set: - for index in random.sample(range(len(train_dataset)), 3): - logger.info(f"Sample {index} of the training set: {train_dataset[index]}.") - - # Get the metric function - if data_args.task_name is not None: - metric = load_metric("glue", data_args.task_name) - # TODO: When datasets metrics include regular accuracy, make an else here and remove special branch from - # compute_metrics - - # You can define your custom compute_metrics function. It takes an `EvalPrediction` object (a namedtuple with a - # predictions and label_ids field) and has to return a dictionary string to float. - def compute_metrics(p: EvalPrediction): - preds = p.predictions[0] if isinstance(p.predictions, tuple) else p.predictions - preds = np.squeeze(preds) if is_regression else np.argmax(preds, axis=1) - if data_args.task_name is not None: - result = metric.compute(predictions=preds, references=p.label_ids) - if len(result) > 1: - result["combined_score"] = np.mean(list(result.values())).item() - return result - elif is_regression: - return {"mse": ((preds - p.label_ids)**2).mean().item()} - else: - return {"accuracy": (preds == p.label_ids).astype(np.float32).mean().item()} - - # Data collator will default to DataCollatorWithPadding, so we change it if we already did the padding. - if data_args.pad_to_max_length: - data_collator = default_data_collator - elif training_args.fp16: - data_collator = DataCollatorWithPadding(tokenizer, pad_to_multiple_of=8) - else: - data_collator = None - - # Initialize our Trainer - trainer = Trainer( - model=model, - args=training_args, - train_dataset=train_dataset, - eval_dataset=eval_dataset if training_args.do_eval else None, - compute_metrics=compute_metrics, - tokenizer=tokenizer, - data_collator=data_collator, - ) - - # Training - if training_args.do_train: - if last_checkpoint is not None: - checkpoint = last_checkpoint - elif os.path.isdir(model_args.model_name_or_path): - checkpoint = model_args.model_name_or_path - else: - checkpoint = None - train_result = trainer.train(resume_from_checkpoint=checkpoint) - metrics = train_result.metrics - trainer.save_model() # Saves the tokenizer too for easy upload - output_train_file = os.path.join(training_args.output_dir, "train_results.txt") - if trainer.is_world_process_zero(): - with open(output_train_file, "w") as writer: - logger.info("***** Train results *****") - for key, value in sorted(metrics.items()): - logger.info(f" {key} = {value}") - writer.write(f"{key} = {value}\n") - - # Need to save the state, since Trainer.save_model saves only the tokenizer with the model - trainer.state.save_to_json( - os.path.join(training_args.output_dir, - "trainer_state.json")) - - # Evaluation - eval_results = {} - if training_args.do_eval: - logger.info("*** Evaluate ***") - - # Loop to handle MNLI double evaluation (matched, mis-matched) - tasks = [data_args.task_name] - eval_datasets = [eval_dataset] - if data_args.task_name == "mnli": - tasks.append("mnli-mm") - eval_datasets.append(datasets["validation_mismatched"]) - - for eval_dataset, task in zip(eval_datasets, tasks): - eval_result = trainer.evaluate(eval_dataset=eval_dataset) - - output_eval_file = os.path.join(training_args.output_dir, - f"eval_results_{task}.txt") - if trainer.is_world_process_zero(): - with open(output_eval_file, "w") as writer: - logger.info(f"***** Eval results {task} *****") - for key, value in sorted(eval_result.items()): - logger.info(f" {key} = {value}") - writer.write(f"{key} = {value}\n") - - eval_results.update(eval_result) - - if training_args.do_predict: - logger.info("*** Test ***") - - # Loop to handle MNLI double evaluation (matched, mis-matched) - tasks = [data_args.task_name] - test_datasets = [test_dataset] - if data_args.task_name == "mnli": - tasks.append("mnli-mm") - test_datasets.append(datasets["test_mismatched"]) - - for test_dataset, task in zip(test_datasets, tasks): - # Removing the `label` columns because it contains -1 and Trainer won't like that. - test_dataset.remove_columns_("label") - predictions = trainer.predict(test_dataset=test_dataset).predictions - predictions = np.squeeze(predictions) if is_regression else np.argmax( - predictions, - axis=1) - - output_test_file = os.path.join(training_args.output_dir, - f"test_results_{task}.txt") - if trainer.is_world_process_zero(): - with open(output_test_file, "w") as writer: - logger.info(f"***** Test results {task} *****") - writer.write("index\tprediction\n") - for index, item in enumerate(predictions): - if is_regression: - writer.write(f"{index}\t{item:3.3f}\n") - else: - item = label_list[item] - writer.write(f"{index}\t{item}\n") - return eval_results - - -def _mp_fn(index): - # For xla_spawn (TPUs) - main() - - -if __name__ == "__main__": - main() diff --git a/training/MoQ/test.json b/training/MoQ/test.json deleted file mode 100644 index a4601fa1d..000000000 --- a/training/MoQ/test.json +++ /dev/null @@ -1,25 +0,0 @@ -{ - "steps_per_print": 10, - "gradient_clipping": 1.0, - "fp16": { - "initial_scale_power": 16, - "enabled": true - }, - "quantize_training": { - "enabled": true, - "quantize_verbose": true, - "quantizer_kernel": true, - "quantize_algo": { - "q_type": "symmetric" - }, - "quantize_bits": { - "start_bits": 16, - "target_bits": 8 - }, - "quantize_schedule": { - "quantize_period": 400, - "schedule_offset": 0 - }, - "quantize_groups": 8 - } -} diff --git a/training/bing_bert/01_adam/mpi_ethernet/deepspeed_bsz4k_01adam_config_seq128_mpi_ethernet.json b/training/bing_bert/01_adam/mpi_ethernet/deepspeed_bsz4k_01adam_config_seq128_mpi_ethernet.json deleted file mode 100644 index a4b50f424..000000000 --- a/training/bing_bert/01_adam/mpi_ethernet/deepspeed_bsz4k_01adam_config_seq128_mpi_ethernet.json +++ /dev/null @@ -1,27 +0,0 @@ -{ - "train_batch_size": 4096, - "train_micro_batch_size_per_gpu": 16, - "steps_per_print": 100, - "prescale_gradients": false, - "optimizer": { - "type": "ZeroOneAdam", - "params": { - "lr": 4e-4, - "weight_decay": 0.01, - "bias_correction": false, - "var_freeze_step": 12500, - "local_step_scaler": 32678, - "cuda_aware": false, - "comm_backend_name": "nccl" - } - }, - "gradient_clipping": 1.0, - - "wall_clock_breakdown": false, - - "fp16": { - "enabled": true, - "loss_scale": 0, - "initial_scale_power": 16 - } -} diff --git a/training/bing_bert/01_adam/mpi_ethernet/deepspeed_bsz4k_01adam_config_seq512_mpi_ethernet.json b/training/bing_bert/01_adam/mpi_ethernet/deepspeed_bsz4k_01adam_config_seq512_mpi_ethernet.json deleted file mode 100644 index 0089970f2..000000000 --- a/training/bing_bert/01_adam/mpi_ethernet/deepspeed_bsz4k_01adam_config_seq512_mpi_ethernet.json +++ /dev/null @@ -1,27 +0,0 @@ -{ - "train_batch_size": 4096, - "train_micro_batch_size_per_gpu": 16, - "steps_per_print": 100, - "prescale_gradients": false, - "optimizer": { - "type": "ZeroOneAdam", - "params": { - "lr": 2.82e-5, - "weight_decay": 0.01, - "bias_correction": false, - "var_freeze_step": 155000, - "local_step_scaler": 32678, - "cuda_aware": false, - "comm_backend_name": "nccl" - } - }, - "gradient_clipping": 1.0, - - "wall_clock_breakdown": false, - - "fp16": { - "enabled": true, - "loss_scale": 0, - "initial_scale_power": 16 - } - } \ No newline at end of file diff --git a/training/bing_bert/01_adam/mpi_ethernet/ds_train_bert_01adam_bsz4k_seq128_mpi_ethernet.sh b/training/bing_bert/01_adam/mpi_ethernet/ds_train_bert_01adam_bsz4k_seq128_mpi_ethernet.sh deleted file mode 100644 index 4b8d43021..000000000 --- a/training/bing_bert/01_adam/mpi_ethernet/ds_train_bert_01adam_bsz4k_seq128_mpi_ethernet.sh +++ /dev/null @@ -1,31 +0,0 @@ -#!/bin/bash - -# This script requires pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs). -# Read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/zero-one-adam/ - -base_dir=`pwd` - -JOB_NAME=01adam_bsz4k_seq128_mpi_ethernet -OUTPUT_DIR=${base_dir}/bert_model_outputs - -mkdir -p $OUTPUT_DIR - -# NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 are used to disable infiniband. Remove it if needed. -run_cmd="NCCL_TREE_THRESHOLD=0 NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 deepspeed --launcher=openmpi \ - ${base_dir}/../../deepspeed_train.py \ - --cf ${base_dir}/../../bert_large.json \ - --max_seq_length 128 \ - --output_dir $OUTPUT_DIR \ - --deepspeed \ - --print_steps 40 \ - --lr_schedule "LE" \ - --lr_offset 0.0 \ - --job_name $JOB_NAME \ - --deepspeed_config ${base_dir}/deepspeed_bsz4k_01adam_config_seq128_mpi_ethernet.json \ - --data_path_prefix /data/bert \ - &> ${JOB_NAME}.log" - -echo ${run_cmd} -eval ${run_cmd} \ No newline at end of file diff --git a/training/bing_bert/01_adam/mpi_ethernet/ds_train_bert_01adam_bsz4k_seq512_mpi_ethernet.sh b/training/bing_bert/01_adam/mpi_ethernet/ds_train_bert_01adam_bsz4k_seq512_mpi_ethernet.sh deleted file mode 100644 index 120f98f94..000000000 --- a/training/bing_bert/01_adam/mpi_ethernet/ds_train_bert_01adam_bsz4k_seq512_mpi_ethernet.sh +++ /dev/null @@ -1,37 +0,0 @@ -#!/bin/bash - -# This script requires pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs). -# Read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/zero-one-adam/ - -base_dir=`pwd` - -JOB_NAME=01adam_bsz4k_seq512_mpi_ethernet -OUTPUT_DIR=${base_dir}/bert_model_outputs -CHECKPOINT_BASE_PATH=${base_dir}/bert_model_outputs/saved_models/01adam_bsz4k_seq128_mpi_ethernet/ -# The default checkpoint to load is the one at the 152K step. -CHECKPOINT_NAME=epoch233_step152511 -echo ${CHECKPOINT_NAME} - -mkdir -p $OUTPUT_DIR - -# NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 are used to disable infiniband. Remove it if needed. -run_cmd="NCCL_TREE_THRESHOLD=0 NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 deepspeed --launcher=openmpi \ - ${base_dir}/../../deepspeed_train.py \ - --cf ${base_dir}/../../bert_large.json \ - --max_seq_length 512 \ - --output_dir $OUTPUT_DIR \ - --deepspeed \ - --print_steps 40 \ - --lr_schedule "LE" \ - --lr_offset 0.0 \ - --job_name $JOB_NAME \ - --deepspeed_config ${base_dir}/deepspeed_bsz4k_01adam_config_seq512_mpi_ethernet.json \ - --data_path_prefix /data/bert \ - --load_training_checkpoint ${CHECKPOINT_BASE_PATH} \ - --load_checkpoint_id ${CHECKPOINT_NAME} \ - &> ${JOB_NAME}.log" - -echo ${run_cmd} -eval ${run_cmd} \ No newline at end of file diff --git a/training/bing_bert/01_adam/mpi_infiniband/deepspeed_bsz4k_01adam_config_seq128_mpi_infiniband.json b/training/bing_bert/01_adam/mpi_infiniband/deepspeed_bsz4k_01adam_config_seq128_mpi_infiniband.json deleted file mode 100644 index a4b50f424..000000000 --- a/training/bing_bert/01_adam/mpi_infiniband/deepspeed_bsz4k_01adam_config_seq128_mpi_infiniband.json +++ /dev/null @@ -1,27 +0,0 @@ -{ - "train_batch_size": 4096, - "train_micro_batch_size_per_gpu": 16, - "steps_per_print": 100, - "prescale_gradients": false, - "optimizer": { - "type": "ZeroOneAdam", - "params": { - "lr": 4e-4, - "weight_decay": 0.01, - "bias_correction": false, - "var_freeze_step": 12500, - "local_step_scaler": 32678, - "cuda_aware": false, - "comm_backend_name": "nccl" - } - }, - "gradient_clipping": 1.0, - - "wall_clock_breakdown": false, - - "fp16": { - "enabled": true, - "loss_scale": 0, - "initial_scale_power": 16 - } -} diff --git a/training/bing_bert/01_adam/mpi_infiniband/deepspeed_bsz4k_01adam_config_seq512_mpi_infiniband.json b/training/bing_bert/01_adam/mpi_infiniband/deepspeed_bsz4k_01adam_config_seq512_mpi_infiniband.json deleted file mode 100644 index 0089970f2..000000000 --- a/training/bing_bert/01_adam/mpi_infiniband/deepspeed_bsz4k_01adam_config_seq512_mpi_infiniband.json +++ /dev/null @@ -1,27 +0,0 @@ -{ - "train_batch_size": 4096, - "train_micro_batch_size_per_gpu": 16, - "steps_per_print": 100, - "prescale_gradients": false, - "optimizer": { - "type": "ZeroOneAdam", - "params": { - "lr": 2.82e-5, - "weight_decay": 0.01, - "bias_correction": false, - "var_freeze_step": 155000, - "local_step_scaler": 32678, - "cuda_aware": false, - "comm_backend_name": "nccl" - } - }, - "gradient_clipping": 1.0, - - "wall_clock_breakdown": false, - - "fp16": { - "enabled": true, - "loss_scale": 0, - "initial_scale_power": 16 - } - } \ No newline at end of file diff --git a/training/bing_bert/01_adam/mpi_infiniband/ds_train_bert_01adam_bsz4k_seq128_mpi_infiniband.sh b/training/bing_bert/01_adam/mpi_infiniband/ds_train_bert_01adam_bsz4k_seq128_mpi_infiniband.sh deleted file mode 100644 index c7761ffeb..000000000 --- a/training/bing_bert/01_adam/mpi_infiniband/ds_train_bert_01adam_bsz4k_seq128_mpi_infiniband.sh +++ /dev/null @@ -1,30 +0,0 @@ -#!/bin/bash - -# This script requires pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs). -# Read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/zero-one-adam/ - -base_dir=`pwd` - -JOB_NAME=01adam_bsz4k_seq128_mpi_infiniband -OUTPUT_DIR=${base_dir}/bert_model_outputs - -mkdir -p $OUTPUT_DIR - -# NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 are used to disable infiniband. Remove it if needed. -run_cmd="NCCL_TREE_THRESHOLD=0 deepspeed --launcher=mvapich ${base_dir}/../../deepspeed_train.py \ - --cf ${base_dir}/../../bert_large.json \ - --max_seq_length 128 \ - --output_dir $OUTPUT_DIR \ - --deepspeed \ - --print_steps 40 \ - --lr_schedule "LE" \ - --lr_offset 0.0 \ - --job_name $JOB_NAME \ - --deepspeed_config ${base_dir}/deepspeed_bsz4k_01adam_config_seq128_mpi_infiniband.json \ - --data_path_prefix /data/bert \ - &> ${JOB_NAME}.log" - -echo ${run_cmd} -eval ${run_cmd} \ No newline at end of file diff --git a/training/bing_bert/01_adam/mpi_infiniband/ds_train_bert_01adam_bsz4k_seq512_mpi_infiniband.sh b/training/bing_bert/01_adam/mpi_infiniband/ds_train_bert_01adam_bsz4k_seq512_mpi_infiniband.sh deleted file mode 100644 index d2395bfd1..000000000 --- a/training/bing_bert/01_adam/mpi_infiniband/ds_train_bert_01adam_bsz4k_seq512_mpi_infiniband.sh +++ /dev/null @@ -1,36 +0,0 @@ -#!/bin/bash - -# This script requires pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs). -# Read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/zero-one-adam/ - -base_dir=`pwd` - -JOB_NAME=01adam_bsz4k_seq512_mpi_infiniband -OUTPUT_DIR=${base_dir}/bert_model_outputs -CHECKPOINT_BASE_PATH=${base_dir}/bert_model_outputs/saved_models/01adam_bsz4k_seq128_mpi_infiniband/ -# The default checkpoint to load is the one at the 152K step. -CHECKPOINT_NAME=epoch233_step152511 -echo ${CHECKPOINT_NAME} - -mkdir -p $OUTPUT_DIR - -# NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 are used to disable infiniband. Remove it if needed. -run_cmd="NCCL_TREE_THRESHOLD=0 deepspeed --launcher=mvapich ${base_dir}/../../deepspeed_train.py \ - --cf ${base_dir}/../../bert_large.json \ - --max_seq_length 512 \ - --output_dir $OUTPUT_DIR \ - --deepspeed \ - --print_steps 40 \ - --lr_schedule "LE" \ - --lr_offset 0.0 \ - --job_name $JOB_NAME \ - --deepspeed_config ${base_dir}/deepspeed_bsz4k_01adam_config_seq512_mpi_infiniband.json \ - --data_path_prefix /data/bert \ - --load_training_checkpoint ${CHECKPOINT_BASE_PATH} \ - --load_checkpoint_id ${CHECKPOINT_NAME} \ - &> ${JOB_NAME}.log" - -echo ${run_cmd} -eval ${run_cmd} \ No newline at end of file diff --git a/training/bing_bert/01_adam/nccl/deepspeed_bsz4k_01adam_config_seq128_nccl.json b/training/bing_bert/01_adam/nccl/deepspeed_bsz4k_01adam_config_seq128_nccl.json deleted file mode 100644 index a4b50f424..000000000 --- a/training/bing_bert/01_adam/nccl/deepspeed_bsz4k_01adam_config_seq128_nccl.json +++ /dev/null @@ -1,27 +0,0 @@ -{ - "train_batch_size": 4096, - "train_micro_batch_size_per_gpu": 16, - "steps_per_print": 100, - "prescale_gradients": false, - "optimizer": { - "type": "ZeroOneAdam", - "params": { - "lr": 4e-4, - "weight_decay": 0.01, - "bias_correction": false, - "var_freeze_step": 12500, - "local_step_scaler": 32678, - "cuda_aware": false, - "comm_backend_name": "nccl" - } - }, - "gradient_clipping": 1.0, - - "wall_clock_breakdown": false, - - "fp16": { - "enabled": true, - "loss_scale": 0, - "initial_scale_power": 16 - } -} diff --git a/training/bing_bert/01_adam/nccl/deepspeed_bsz4k_01adam_config_seq512_nccl.json b/training/bing_bert/01_adam/nccl/deepspeed_bsz4k_01adam_config_seq512_nccl.json deleted file mode 100644 index 0089970f2..000000000 --- a/training/bing_bert/01_adam/nccl/deepspeed_bsz4k_01adam_config_seq512_nccl.json +++ /dev/null @@ -1,27 +0,0 @@ -{ - "train_batch_size": 4096, - "train_micro_batch_size_per_gpu": 16, - "steps_per_print": 100, - "prescale_gradients": false, - "optimizer": { - "type": "ZeroOneAdam", - "params": { - "lr": 2.82e-5, - "weight_decay": 0.01, - "bias_correction": false, - "var_freeze_step": 155000, - "local_step_scaler": 32678, - "cuda_aware": false, - "comm_backend_name": "nccl" - } - }, - "gradient_clipping": 1.0, - - "wall_clock_breakdown": false, - - "fp16": { - "enabled": true, - "loss_scale": 0, - "initial_scale_power": 16 - } - } \ No newline at end of file diff --git a/training/bing_bert/01_adam/nccl/ds_train_bert_01adam_bsz4k_seq128_nccl.sh b/training/bing_bert/01_adam/nccl/ds_train_bert_01adam_bsz4k_seq128_nccl.sh deleted file mode 100644 index a631c85c6..000000000 --- a/training/bing_bert/01_adam/nccl/ds_train_bert_01adam_bsz4k_seq128_nccl.sh +++ /dev/null @@ -1,32 +0,0 @@ -#!/bin/bash - -# This script requires pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs). -# Read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/zero-one-adam/ - -base_dir=`pwd` - -JOB_NAME=01adam_bsz4k_seq128_nccl -OUTPUT_DIR=${base_dir}/bert_model_outputs - -mkdir -p $OUTPUT_DIR - -# NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 are used to disable infiniband. Remove it if needed. -run_cmd="NCCL_TREE_THRESHOLD=0 NCCL_DEBUG=INFO \ - deepspeed \ - ${base_dir}/../../deepspeed_train.py \ - --cf ${base_dir}/../../bert_large.json \ - --max_seq_length 128 \ - --output_dir $OUTPUT_DIR \ - --deepspeed \ - --print_steps 40 \ - --lr_schedule "LE" \ - --lr_offset 0.0 \ - --job_name $JOB_NAME \ - --deepspeed_config ${base_dir}/deepspeed_bsz4k_01adam_config_seq128_nccl.json \ - --data_path_prefix /data/bert \ - &> ${JOB_NAME}.log" - -echo ${run_cmd} -eval ${run_cmd} \ No newline at end of file diff --git a/training/bing_bert/01_adam/nccl/ds_train_bert_01adam_bsz4k_seq512_nccl.sh b/training/bing_bert/01_adam/nccl/ds_train_bert_01adam_bsz4k_seq512_nccl.sh deleted file mode 100644 index 06a084af3..000000000 --- a/training/bing_bert/01_adam/nccl/ds_train_bert_01adam_bsz4k_seq512_nccl.sh +++ /dev/null @@ -1,38 +0,0 @@ -#!/bin/bash - -# This script requires pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs). -# Read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/zero-one-adam/ - -base_dir=`pwd` - -JOB_NAME=01adam_bsz4k_seq512_nccl -OUTPUT_DIR=${base_dir}/bert_model_outputs -CHECKPOINT_BASE_PATH=${base_dir}/bert_model_outputs/saved_models/01adam_bsz4k_seq128_nccl/ -# The default checkpoint to load is the one at the 152K step. -CHECKPOINT_NAME=epoch233_step152511 -echo ${CHECKPOINT_NAME} - -mkdir -p $OUTPUT_DIR - -# NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 are used to disable infiniband. Remove it if needed. -run_cmd="NCCL_TREE_THRESHOLD=0 NCCL_DEBUG=INFO \ - deepspeed \ - ${base_dir}/../../deepspeed_train.py \ - --cf ${base_dir}/../../bert_large.json \ - --max_seq_length 512 \ - --output_dir $OUTPUT_DIR \ - --deepspeed \ - --print_steps 40 \ - --lr_schedule "LE" \ - --lr_offset 0.0 \ - --job_name $JOB_NAME \ - --deepspeed_config ${base_dir}/deepspeed_bsz4k_01adam_config_seq512_nccl.json \ - --data_path_prefix /data/bert \ - --load_training_checkpoint ${CHECKPOINT_BASE_PATH} \ - --load_checkpoint_id ${CHECKPOINT_NAME} \ - &> ${JOB_NAME}.log" - -echo ${run_cmd} -eval ${run_cmd} \ No newline at end of file diff --git a/training/bing_bert/1-bit_adam/mpi_ethernet/deepspeed_bsz4k_onebitadam_config_seq128_mpi_ethernet.json b/training/bing_bert/1-bit_adam/mpi_ethernet/deepspeed_bsz4k_onebitadam_config_seq128_mpi_ethernet.json deleted file mode 100644 index 49d463828..000000000 --- a/training/bing_bert/1-bit_adam/mpi_ethernet/deepspeed_bsz4k_onebitadam_config_seq128_mpi_ethernet.json +++ /dev/null @@ -1,26 +0,0 @@ -{ - "train_batch_size": 4096, - "train_micro_batch_size_per_gpu": 16, - "steps_per_print": 100, - "prescale_gradients": false, - "optimizer": { - "type": "OneBitAdam", - "params": { - "lr": 4e-4, - "weight_decay": 0.01, - "bias_correction": false, - "freeze_step": 23000, - "cuda_aware": false, - "comm_backend_name": "mpi" - } - }, - "gradient_clipping": 1.0, - - "wall_clock_breakdown": false, - - "fp16": { - "enabled": true, - "loss_scale": 0, - "initial_scale_power": 16 - } -} diff --git a/training/bing_bert/1-bit_adam/mpi_ethernet/ds_train_bert_onebitadam_bsz4k_seq128_mpi_ethernet.sh b/training/bing_bert/1-bit_adam/mpi_ethernet/ds_train_bert_onebitadam_bsz4k_seq128_mpi_ethernet.sh deleted file mode 100644 index 6a0c9bda8..000000000 --- a/training/bing_bert/1-bit_adam/mpi_ethernet/ds_train_bert_onebitadam_bsz4k_seq128_mpi_ethernet.sh +++ /dev/null @@ -1,33 +0,0 @@ -#!/bin/bash - -# If you are able to install pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs), -# we highly recommend you to use the NCCL-based 1-bit Adam -# which has better performance and ease of use -# (see scripts in DeepSpeedExamples/bing_bert/1-bit_adam/nccl -# and read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-adam/) - -base_dir=`pwd` - -# Where should we save checkpoints and tensorboard events? -JOB_NAME=onebit_adam_4k_seq128_mpi_ethernet -OUTPUT_DIR=${base_dir}/bert_model_outputs - -mkdir -p $OUTPUT_DIR - -# NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 are used to disable infiniband. Remove it if needed. -NCCL_TREE_THRESHOLD=0 NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 deepspeed --launcher=openmpi ${base_dir}/../../deepspeed_train.py \ ---cf ${base_dir}/../../bert_large.json \ ---max_seq_length 128 \ ---output_dir $OUTPUT_DIR \ ---deepspeed_mpi \ ---deepspeed \ ---deepspeed_transformer_kernel \ ---print_steps 40 \ ---lr_schedule "LE" \ ---lr_offset 0.0 \ ---job_name $JOB_NAME \ ---deepspeed_config ${base_dir}/deepspeed_bsz4k_onebitadam_config_seq128_mpi_ethernet.json \ ---data_path_prefix /data/bert \ -&> ${JOB_NAME}.log diff --git a/training/bing_bert/1-bit_adam/mpi_ethernet/mpi_train_bert_onebitadam_bsz4k_seq128_ethernet.sh b/training/bing_bert/1-bit_adam/mpi_ethernet/mpi_train_bert_onebitadam_bsz4k_seq128_ethernet.sh deleted file mode 100644 index a90691655..000000000 --- a/training/bing_bert/1-bit_adam/mpi_ethernet/mpi_train_bert_onebitadam_bsz4k_seq128_ethernet.sh +++ /dev/null @@ -1,39 +0,0 @@ -#!/bin/bash - -# If you are able to install pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs), -# we highly recommend you to use the NCCL-based 1-bit Adam -# which has better performance and ease of use -# (see scripts in DeepSpeedExamples/bing_bert/1-bit_adam/nccl -# and read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-adam/) - -# Note: Please use the deepspeed launch script for most cases. -# For advanced users, mpirun or other MPI launchers can be used -# with this script as follows. -# mpirun -n 2 [launcher-args] python ${base_dir}/../../deepspeed_train.py -# As an example, below we include the command we used - -base_dir=`pwd` - -# Where should we save checkpoints and tensorboard events? -JOB_NAME=onebit_adam_4k_seq128_mpirun_ethernet -OUTPUT_DIR=${base_dir}/bert_model_outputs - -mkdir -p $OUTPUT_DIR - -# NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 are used to disable infiniband. Remove it if needed. -mpirun -n 128 -npernode 4 -hostfile /job/hostfile -x UCX_TLS=tcp --mca btl ^openib --mca btl_tcp_if_include eth0 -x NCCL_TREE_THRESHOLD=0 -x NCCL_IB_DISABLE=1 -x NCCL_SOCKET_IFNAME=eth0 python ${base_dir}/../../deepspeed_train.py \ ---cf ${base_dir}/../../bert_large.json \ ---max_seq_length 128 \ ---output_dir $OUTPUT_DIR \ ---deepspeed_mpi \ ---deepspeed \ ---deepspeed_transformer_kernel \ ---print_steps 40 \ ---lr_schedule "LE" \ ---lr_offset 0.0 \ ---job_name $JOB_NAME \ ---deepspeed_config ${base_dir}/deepspeed_bsz4k_onebitadam_config_seq128_mpi_ethernet.json \ ---data_path_prefix /data/bert \ -&> ${JOB_NAME}.log diff --git a/training/bing_bert/1-bit_adam/mpi_infiniband/deepspeed_bsz4k_onebitadam_config_seq128_mpi_infiniband.json b/training/bing_bert/1-bit_adam/mpi_infiniband/deepspeed_bsz4k_onebitadam_config_seq128_mpi_infiniband.json deleted file mode 100644 index 9963b3c7d..000000000 --- a/training/bing_bert/1-bit_adam/mpi_infiniband/deepspeed_bsz4k_onebitadam_config_seq128_mpi_infiniband.json +++ /dev/null @@ -1,26 +0,0 @@ -{ - "train_batch_size": 4096, - "train_micro_batch_size_per_gpu": 16, - "steps_per_print": 100, - "prescale_gradients": false, - "optimizer": { - "type": "OneBitAdam", - "params": { - "lr": 4e-4, - "weight_decay": 0.01, - "bias_correction": false, - "freeze_step": 23000, - "cuda_aware": true, - "comm_backend_name": "mpi" - } - }, - "gradient_clipping": 1.0, - - "wall_clock_breakdown": false, - - "fp16": { - "enabled": true, - "loss_scale": 0, - "initial_scale_power": 16 - } -} diff --git a/training/bing_bert/1-bit_adam/mpi_infiniband/ds_train_bert_onebitadam_bsz4k_seq128_mpi_infiniband.sh b/training/bing_bert/1-bit_adam/mpi_infiniband/ds_train_bert_onebitadam_bsz4k_seq128_mpi_infiniband.sh deleted file mode 100644 index 32420cffe..000000000 --- a/training/bing_bert/1-bit_adam/mpi_infiniband/ds_train_bert_onebitadam_bsz4k_seq128_mpi_infiniband.sh +++ /dev/null @@ -1,32 +0,0 @@ -#!/bin/bash - -# If you are able to install pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs), -# we highly recommend you to use the NCCL-based 1-bit Adam -# which has better performance and ease of use -# (see scripts in DeepSpeedExamples/bing_bert/1-bit_adam/nccl -# and read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-adam/) - -base_dir=`pwd` - -# Where should we save checkpoints and tensorboard events? -JOB_NAME=onebit_adam_4k_seq128_mpi_infiniband -OUTPUT_DIR=${base_dir}/bert_model_outputs - -mkdir -p $OUTPUT_DIR - -NCCL_TREE_THRESHOLD=0 deepspeed --launcher=mvapich ${base_dir}/../../deepspeed_train.py \ ---cf ${base_dir}/../../bert_large.json \ ---max_seq_length 128 \ ---output_dir $OUTPUT_DIR \ ---deepspeed_mpi \ ---deepspeed \ ---deepspeed_transformer_kernel \ ---print_steps 40 \ ---lr_schedule "LE" \ ---lr_offset 0.0 \ ---job_name $JOB_NAME \ ---deepspeed_config ${base_dir}/deepspeed_bsz4k_onebitadam_config_seq128_mpi_infiniband.json \ ---data_path_prefix /data/bert \ -&> ${JOB_NAME}.log diff --git a/training/bing_bert/1-bit_adam/mpi_infiniband/mpi_train_bert_onebitadam_bsz4k_seq128_infiniband.sh b/training/bing_bert/1-bit_adam/mpi_infiniband/mpi_train_bert_onebitadam_bsz4k_seq128_infiniband.sh deleted file mode 100644 index a8976a403..000000000 --- a/training/bing_bert/1-bit_adam/mpi_infiniband/mpi_train_bert_onebitadam_bsz4k_seq128_infiniband.sh +++ /dev/null @@ -1,38 +0,0 @@ -#!/bin/bash - -# If you are able to install pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs), -# we highly recommend you to use the NCCL-based 1-bit Adam -# which has better performance and ease of use -# (see scripts in DeepSpeedExamples/bing_bert/1-bit_adam/nccl -# and read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-adam/) - -# Note: Please use the deepspeed launch script for most cases. -# For advanced users, mpirun or other MPI launchers can be used -# with this script as follows. -# mpirun -n 2 [launcher-args] python ${base_dir}/../../deepspeed_train.py -# As an example, below we include the command we used - -base_dir=`pwd` - -# Where should we save checkpoints and tensorboard events? -JOB_NAME=onebit_adam_4k_seq128_mpirun_infiniband -OUTPUT_DIR=${base_dir}/bert_model_outputs - -mkdir -p $OUTPUT_DIR - -mpirun -n 128 -ppn 8 -f /tmp/deepspeed_mvapich_hostfile -env MV2_SUPPORT_DL=1 -env MV2_USE_GDR=0 -env MV2_USE_CUDA=1 -env MV2_USE_GDRCOPY=0 -env MV2_SMP_USE_CMA=0 -env MV2_DEBUG_SHOW_BACKTRACE=1 python ${base_dir}/../../deepspeed_train.py \ ---cf ${base_dir}/../../bert_large.json \ ---max_seq_length 128 \ ---output_dir $OUTPUT_DIR \ ---deepspeed_mpi \ ---deepspeed \ ---deepspeed_transformer_kernel \ ---print_steps 40 \ ---lr_schedule "LE" \ ---lr_offset 0.0 \ ---job_name $JOB_NAME \ ---deepspeed_config ${base_dir}/deepspeed_bsz4k_onebitadam_config_seq128_mpi_infiniband.json \ ---data_path_prefix /data/bert \ -&> ${JOB_NAME}.log diff --git a/training/bing_bert/1-bit_adam/nccl/deepspeed_bsz4k_onebitadam_config_seq128_nccl.json b/training/bing_bert/1-bit_adam/nccl/deepspeed_bsz4k_onebitadam_config_seq128_nccl.json deleted file mode 100644 index 9f7109b00..000000000 --- a/training/bing_bert/1-bit_adam/nccl/deepspeed_bsz4k_onebitadam_config_seq128_nccl.json +++ /dev/null @@ -1,26 +0,0 @@ -{ - "train_batch_size": 4096, - "train_micro_batch_size_per_gpu": 16, - "steps_per_print": 100, - "prescale_gradients": false, - "optimizer": { - "type": "OneBitAdam", - "params": { - "lr": 4e-4, - "weight_decay": 0.01, - "bias_correction": false, - "freeze_step": 23000, - "cuda_aware": false, - "comm_backend_name": "nccl" - } - }, - "gradient_clipping": 1.0, - - "wall_clock_breakdown": false, - - "fp16": { - "enabled": true, - "loss_scale": 0, - "initial_scale_power": 16 - } -} diff --git a/training/bing_bert/1-bit_adam/nccl/ds_train_bert_onebitadam_bsz4k_seq128_nccl.sh b/training/bing_bert/1-bit_adam/nccl/ds_train_bert_onebitadam_bsz4k_seq128_nccl.sh deleted file mode 100644 index b6fb5a644..000000000 --- a/training/bing_bert/1-bit_adam/nccl/ds_train_bert_onebitadam_bsz4k_seq128_nccl.sh +++ /dev/null @@ -1,29 +0,0 @@ -#!/bin/bash - -# This script requires pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs). -# Read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-adam/ - -base_dir=`pwd` - -# Where should we save checkpoints and tensorboard events? -JOB_NAME=onebit_adam_4k_seq128_nccl -OUTPUT_DIR=${base_dir}/bert_model_outputs - -mkdir -p $OUTPUT_DIR - -# NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 are used to disable infiniband. Remove it if needed. -NCCL_TREE_THRESHOLD=0 NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 deepspeed ${base_dir}/../../deepspeed_train.py \ ---cf ${base_dir}/../../bert_large.json \ ---max_seq_length 128 \ ---output_dir $OUTPUT_DIR \ ---deepspeed \ ---deepspeed_transformer_kernel \ ---print_steps 40 \ ---lr_schedule "LE" \ ---lr_offset 0.0 \ ---job_name $JOB_NAME \ ---deepspeed_config ${base_dir}/deepspeed_bsz4k_onebitadam_config_seq128_nccl.json \ ---data_path_prefix /data/bert \ -&> ${JOB_NAME}.log diff --git a/training/bing_bert/1-bit_lamb/mpi_ethernet/deepspeed_bsz32k_onebitlamb_config_seq512_mpi_ethernet.json b/training/bing_bert/1-bit_lamb/mpi_ethernet/deepspeed_bsz32k_onebitlamb_config_seq512_mpi_ethernet.json deleted file mode 100644 index db24b09f1..000000000 --- a/training/bing_bert/1-bit_lamb/mpi_ethernet/deepspeed_bsz32k_onebitlamb_config_seq512_mpi_ethernet.json +++ /dev/null @@ -1,31 +0,0 @@ -{ - "train_batch_size": 32768, - "train_micro_batch_size_per_gpu": 16, - "steps_per_print": 1000, - "prescale_gradients": false, - "optimizer": { - "type": "OneBitLamb", - "params": { - "lr": 2e-3, - "weight_decay": 0.01, - "bias_correction": false, - "max_coeff": 0.3, - "min_coeff": 0.01, - "freeze_step": 6100, - "cuda_aware": false, - "comm_backend_name": "mpi", - "coeff_beta": 0.9, - "factor_max": 4.0, - "factor_min": 0.5, - "factor_threshold": 0.1 - } - }, - "gradient_clipping": 1.0, - - "wall_clock_breakdown": false, - - "fp16": { - "enabled": true, - "loss_scale": 0 - } -} diff --git a/training/bing_bert/1-bit_lamb/mpi_ethernet/deepspeed_bsz64k_onebitlamb_config_seq128_mpi_ethernet.json b/training/bing_bert/1-bit_lamb/mpi_ethernet/deepspeed_bsz64k_onebitlamb_config_seq128_mpi_ethernet.json deleted file mode 100644 index c0e6abaf6..000000000 --- a/training/bing_bert/1-bit_lamb/mpi_ethernet/deepspeed_bsz64k_onebitlamb_config_seq128_mpi_ethernet.json +++ /dev/null @@ -1,32 +0,0 @@ -{ - "train_batch_size": 65536, - "train_micro_batch_size_per_gpu": 64, - "steps_per_print": 1000, - "prescale_gradients": false, - "optimizer": { - "type": "OneBitLamb", - "params": { - "lr": 11e-3, - "weight_decay": 0.01, - "bias_correction": false, - "max_coeff": 0.3, - "min_coeff": 0.01, - "freeze_step": 1000, - "cuda_aware": false, - "comm_backend_name": "mpi", - "coeff_beta": 0.9, - "factor_max": 4.0, - "factor_min": 0.5, - "factor_threshold": 0.1 - } - }, - "gradient_clipping": 1.0, - - "wall_clock_breakdown": false, - - "fp16": { - "enabled": true, - "loss_scale": 0, - "initial_scale_power": 16 - } -} diff --git a/training/bing_bert/1-bit_lamb/mpi_ethernet/ds_train_bert_onebitlamb_bsz32k_seq512_mpi_ethernet.sh b/training/bing_bert/1-bit_lamb/mpi_ethernet/ds_train_bert_onebitlamb_bsz32k_seq512_mpi_ethernet.sh deleted file mode 100644 index 78b0bc303..000000000 --- a/training/bing_bert/1-bit_lamb/mpi_ethernet/ds_train_bert_onebitlamb_bsz32k_seq512_mpi_ethernet.sh +++ /dev/null @@ -1,45 +0,0 @@ -#!/bin/bash - -# If you are able to install pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs), -# we highly recommend you to use the NCCL-based 1-bit Lamb -# which has better performance and ease of use -# (see scripts in DeepSpeedExamples/bing_bert/1-bit_lamb/nccl -# and read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-lamb/) - -base_dir=`pwd` - -# Assumes job name in previous seq128 run, will resume training from epoch 150 -EPOCH=150 - -# Where should we save checkpoints and tensorboard events? -JOB_NAME=onebit_lamb_32k_chkpt${EPOCH}_seq512_mpi_ethernet -OUTPUT_DIR=${base_dir}/bert_model_outputs - -CHECKPOINT_BASE_PATH=${OUTPUT_DIR}/saved_models/onebit_lamb_64k_seq128_mpi_ethernet -CHECKPOINT_NAME=`basename ${CHECKPOINT_BASE_PATH}/epoch${EPOCH}_*` -echo "checkpoint id: $CHECKPOINT_NAME" - -mkdir -p $OUTPUT_DIR - -# NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 are used to disable infiniband. Remove it if needed. -NCCL_TREE_THRESHOLD=0 NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 deepspeed --launcher=openmpi ${base_dir}/../../deepspeed_train.py \ ---cf ${base_dir}/../../bert_large_lamb.json \ ---max_seq_length 512 \ ---output_dir $OUTPUT_DIR \ ---print_steps 100 \ ---deepspeed_mpi \ ---deepspeed \ ---deepspeed_transformer_kernel \ ---job_name $JOB_NAME \ ---deepspeed_config ${base_dir}/deepspeed_bsz32k_onebitlamb_config_seq512_mpi_ethernet.json \ ---data_path_prefix /data/bert \ ---validation_data_path_prefix /data/bert \ ---rewarmup \ ---lr_schedule "EE" \ ---attention_dropout_checkpoint \ ---lr_offset 0.0 \ ---load_training_checkpoint ${CHECKPOINT_BASE_PATH} \ ---load_checkpoint_id ${CHECKPOINT_NAME} \ -&> ${JOB_NAME}.log diff --git a/training/bing_bert/1-bit_lamb/mpi_ethernet/ds_train_bert_onebitlamb_bsz64k_seq128_mpi_ethernet.sh b/training/bing_bert/1-bit_lamb/mpi_ethernet/ds_train_bert_onebitlamb_bsz64k_seq128_mpi_ethernet.sh deleted file mode 100644 index 79ca855d1..000000000 --- a/training/bing_bert/1-bit_lamb/mpi_ethernet/ds_train_bert_onebitlamb_bsz64k_seq128_mpi_ethernet.sh +++ /dev/null @@ -1,33 +0,0 @@ -#!/bin/bash - -# If you are able to install pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs), -# we highly recommend you to use the NCCL-based 1-bit Lamb -# which has better performance and ease of use -# (see scripts in DeepSpeedExamples/bing_bert/1-bit_lamb/nccl -# and read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-lamb/) - -base_dir=`pwd` - -# Where should we save checkpoints and tensorboard events? -JOB_NAME=onebit_lamb_64k_seq128_mpi_ethernet -OUTPUT_DIR=${base_dir}/bert_model_outputs - -mkdir -p $OUTPUT_DIR - -# NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 are used to disable infiniband. Remove it if needed. -NCCL_TREE_THRESHOLD=0 NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 deepspeed --launcher=openmpi ${base_dir}/../../deepspeed_train.py \ ---cf ${base_dir}/../../bert_large_lamb.json \ ---max_seq_length 128 \ ---output_dir $OUTPUT_DIR \ ---deepspeed_mpi \ ---deepspeed \ ---deepspeed_transformer_kernel \ ---print_steps 40 \ ---lr_schedule "EE" \ ---lr_offset 10e-4 \ ---job_name $JOB_NAME \ ---deepspeed_config ${base_dir}/deepspeed_bsz64k_onebitlamb_config_seq128_mpi_ethernet.json \ ---data_path_prefix /data/bert \ -&> ${JOB_NAME}.log diff --git a/training/bing_bert/1-bit_lamb/mpi_ethernet/mpi_train_bert_onebitlamb_bsz32k_seq512_ethernet.sh b/training/bing_bert/1-bit_lamb/mpi_ethernet/mpi_train_bert_onebitlamb_bsz32k_seq512_ethernet.sh deleted file mode 100644 index 3c8d0da03..000000000 --- a/training/bing_bert/1-bit_lamb/mpi_ethernet/mpi_train_bert_onebitlamb_bsz32k_seq512_ethernet.sh +++ /dev/null @@ -1,51 +0,0 @@ -#!/bin/bash - -# If you are able to install pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs), -# we highly recommend you to use the NCCL-based 1-bit Lamb -# which has better performance and ease of use -# (see scripts in DeepSpeedExamples/bing_bert/1-bit_lamb/nccl -# and read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-lamb/) - -# Note: Please use the deepspeed launch script for most cases. -# For advanced users, mpirun or other MPI launchers can be used -# with this script as follows. -# mpirun -n 2 [launcher-args] python ${base_dir}/../deepspeed_train.py -# As an example, below we include the command we used - -base_dir=`pwd` - -# Assumes job name in previous seq128 run, will resume training from epoch 150 -EPOCH=150 - -# Where should we save checkpoints and tensorboard events? -JOB_NAME=onebit_lamb_32k_chkpt${EPOCH}_seq512_mpirun_ethernet -OUTPUT_DIR=${base_dir}/bert_model_outputs - -CHECKPOINT_BASE_PATH=${OUTPUT_DIR}/saved_models/onebit_lamb_64k_seq128_mpirun_ethernet -CHECKPOINT_NAME=`basename ${CHECKPOINT_BASE_PATH}/epoch${EPOCH}_*` -echo "checkpoint id: $CHECKPOINT_NAME" - -mkdir -p $OUTPUT_DIR - -# NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 are used to disable infiniband. Remove it if needed. -mpirun -n 128 -npernode 4 -hostfile /job/hostfile -x UCX_TLS=tcp --mca btl ^openib --mca btl_tcp_if_include eth0 -x NCCL_TREE_THRESHOLD=0 -x NCCL_IB_DISABLE=1 -x NCCL_SOCKET_IFNAME=eth0 python ${base_dir}/../../deepspeed_train.py \ ---cf ${base_dir}/../../bert_large_lamb.json \ ---max_seq_length 512 \ ---output_dir $OUTPUT_DIR \ ---print_steps 100 \ ---deepspeed_mpi \ ---deepspeed \ ---deepspeed_transformer_kernel \ ---job_name $JOB_NAME \ ---deepspeed_config ${base_dir}/deepspeed_bsz32k_onebitlamb_config_seq512_mpi_ethernet.json \ ---data_path_prefix /data/bert \ ---validation_data_path_prefix /data/bert \ ---rewarmup \ ---lr_schedule "EE" \ ---attention_dropout_checkpoint \ ---lr_offset 0.0 \ ---load_training_checkpoint ${CHECKPOINT_BASE_PATH} \ ---load_checkpoint_id ${CHECKPOINT_NAME} \ -&> ${JOB_NAME}.log diff --git a/training/bing_bert/1-bit_lamb/mpi_ethernet/mpi_train_bert_onebitlamb_bsz64k_seq128_ethernet.sh b/training/bing_bert/1-bit_lamb/mpi_ethernet/mpi_train_bert_onebitlamb_bsz64k_seq128_ethernet.sh deleted file mode 100644 index 348f3b8b5..000000000 --- a/training/bing_bert/1-bit_lamb/mpi_ethernet/mpi_train_bert_onebitlamb_bsz64k_seq128_ethernet.sh +++ /dev/null @@ -1,39 +0,0 @@ -#!/bin/bash - -# If you are able to install pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs), -# we highly recommend you to use the NCCL-based 1-bit Lamb -# which has better performance and ease of use -# (see scripts in DeepSpeedExamples/bing_bert/1-bit_lamb/nccl -# and read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-lamb/) - -# Note: Please use the deepspeed launch script for most cases. -# For advanced users, mpirun or other MPI launchers can be used -# with this script as follows. -# mpirun -n 2 [launcher-args] python ${base_dir}/../deepspeed_train.py -# As an example, below we include the command we used - -base_dir=`pwd` - -# Where should we save checkpoints and tensorboard events? -JOB_NAME=onebit_lamb_64k_seq128_mpirun_ethernet -OUTPUT_DIR=${base_dir}/bert_model_outputs - -mkdir -p $OUTPUT_DIR - -# NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 are used to disable infiniband. Remove it if needed. -mpirun -n 128 -npernode 4 -hostfile /job/hostfile -x UCX_TLS=tcp --mca btl ^openib --mca btl_tcp_if_include eth0 -x NCCL_TREE_THRESHOLD=0 -x NCCL_IB_DISABLE=1 -x NCCL_SOCKET_IFNAME=eth0 python ${base_dir}/../../deepspeed_train.py \ ---cf ${base_dir}/../../bert_large_lamb.json \ ---max_seq_length 128 \ ---output_dir $OUTPUT_DIR \ ---deepspeed_mpi \ ---deepspeed \ ---deepspeed_transformer_kernel \ ---print_steps 40 \ ---lr_schedule "EE" \ ---lr_offset 10e-4 \ ---job_name $JOB_NAME \ ---deepspeed_config ${base_dir}/deepspeed_bsz64k_onebitlamb_config_seq128_mpi_ethernet.json \ ---data_path_prefix /data/bert \ -&> ${JOB_NAME}.log diff --git a/training/bing_bert/1-bit_lamb/mpi_infiniband/deepspeed_bsz32k_onebitlamb_config_seq512_mpi_infiniband.json b/training/bing_bert/1-bit_lamb/mpi_infiniband/deepspeed_bsz32k_onebitlamb_config_seq512_mpi_infiniband.json deleted file mode 100644 index 881b1c78c..000000000 --- a/training/bing_bert/1-bit_lamb/mpi_infiniband/deepspeed_bsz32k_onebitlamb_config_seq512_mpi_infiniband.json +++ /dev/null @@ -1,31 +0,0 @@ -{ - "train_batch_size": 32768, - "train_micro_batch_size_per_gpu": 16, - "steps_per_print": 1000, - "prescale_gradients": false, - "optimizer": { - "type": "OneBitLamb", - "params": { - "lr": 2e-3, - "weight_decay": 0.01, - "bias_correction": false, - "max_coeff": 0.3, - "min_coeff": 0.01, - "freeze_step": 6100, - "cuda_aware": true, - "comm_backend_name": "mpi", - "coeff_beta": 0.9, - "factor_max": 4.0, - "factor_min": 0.5, - "factor_threshold": 0.1 - } - }, - "gradient_clipping": 1.0, - - "wall_clock_breakdown": false, - - "fp16": { - "enabled": true, - "loss_scale": 0 - } -} diff --git a/training/bing_bert/1-bit_lamb/mpi_infiniband/deepspeed_bsz64k_onebitlamb_config_seq128_mpi_infiniband.json b/training/bing_bert/1-bit_lamb/mpi_infiniband/deepspeed_bsz64k_onebitlamb_config_seq128_mpi_infiniband.json deleted file mode 100644 index 82683b22c..000000000 --- a/training/bing_bert/1-bit_lamb/mpi_infiniband/deepspeed_bsz64k_onebitlamb_config_seq128_mpi_infiniband.json +++ /dev/null @@ -1,32 +0,0 @@ -{ - "train_batch_size": 65536, - "train_micro_batch_size_per_gpu": 64, - "steps_per_print": 1000, - "prescale_gradients": false, - "optimizer": { - "type": "OneBitLamb", - "params": { - "lr": 11e-3, - "weight_decay": 0.01, - "bias_correction": false, - "max_coeff": 0.3, - "min_coeff": 0.01, - "freeze_step": 1000, - "cuda_aware": true, - "comm_backend_name": "mpi", - "coeff_beta": 0.9, - "factor_max": 4.0, - "factor_min": 0.5, - "factor_threshold": 0.1 - } - }, - "gradient_clipping": 1.0, - - "wall_clock_breakdown": false, - - "fp16": { - "enabled": true, - "loss_scale": 0, - "initial_scale_power": 16 - } -} diff --git a/training/bing_bert/1-bit_lamb/mpi_infiniband/ds_train_bert_onebitlamb_bsz32k_seq512_mpi_infiniband.sh b/training/bing_bert/1-bit_lamb/mpi_infiniband/ds_train_bert_onebitlamb_bsz32k_seq512_mpi_infiniband.sh deleted file mode 100644 index e07b2ecdc..000000000 --- a/training/bing_bert/1-bit_lamb/mpi_infiniband/ds_train_bert_onebitlamb_bsz32k_seq512_mpi_infiniband.sh +++ /dev/null @@ -1,44 +0,0 @@ -#!/bin/bash - -# If you are able to install pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs), -# we highly recommend you to use the NCCL-based 1-bit Lamb -# which has better performance and ease of use -# (see scripts in DeepSpeedExamples/bing_bert/1-bit_lamb/nccl -# and read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-lamb/) - -base_dir=`pwd` - -# Assumes job name in previous seq128 run, will resume training from epoch 150 -EPOCH=150 - -# Where should we save checkpoints and tensorboard events? -JOB_NAME=onebit_lamb_32k_chkpt${EPOCH}_seq512_mpi_infiniband -OUTPUT_DIR=${base_dir}/bert_model_outputs - -CHECKPOINT_BASE_PATH=${OUTPUT_DIR}/saved_models/onebit_lamb_64k_seq128_mpi_infiniband -CHECKPOINT_NAME=`basename ${CHECKPOINT_BASE_PATH}/epoch${EPOCH}_*` -echo "checkpoint id: $CHECKPOINT_NAME" - -mkdir -p $OUTPUT_DIR - -NCCL_TREE_THRESHOLD=0 deepspeed --launcher=mvapich ${base_dir}/../../deepspeed_train.py \ ---cf ${base_dir}/../../bert_large_lamb.json \ ---max_seq_length 512 \ ---output_dir $OUTPUT_DIR \ ---print_steps 100 \ ---deepspeed_mpi \ ---deepspeed \ ---deepspeed_transformer_kernel \ ---job_name $JOB_NAME \ ---deepspeed_config ${base_dir}/deepspeed_bsz32k_onebitlamb_config_seq512_mpi_infiniband.json \ ---data_path_prefix /data/bert \ ---validation_data_path_prefix /data/bert \ ---rewarmup \ ---lr_schedule "EE" \ ---attention_dropout_checkpoint \ ---lr_offset 0.0 \ ---load_training_checkpoint ${CHECKPOINT_BASE_PATH} \ ---load_checkpoint_id ${CHECKPOINT_NAME} \ -&> ${JOB_NAME}.log diff --git a/training/bing_bert/1-bit_lamb/mpi_infiniband/ds_train_bert_onebitlamb_bsz64k_seq128_mpi_infiniband.sh b/training/bing_bert/1-bit_lamb/mpi_infiniband/ds_train_bert_onebitlamb_bsz64k_seq128_mpi_infiniband.sh deleted file mode 100644 index 21890af6c..000000000 --- a/training/bing_bert/1-bit_lamb/mpi_infiniband/ds_train_bert_onebitlamb_bsz64k_seq128_mpi_infiniband.sh +++ /dev/null @@ -1,32 +0,0 @@ -#!/bin/bash - -# If you are able to install pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs), -# we highly recommend you to use the NCCL-based 1-bit Lamb -# which has better performance and ease of use -# (see scripts in DeepSpeedExamples/bing_bert/1-bit_lamb/nccl -# and read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-lamb/) - -base_dir=`pwd` - -# Where should we save checkpoints and tensorboard events? -JOB_NAME=onebit_lamb_64k_seq128_mpi_infiniband -OUTPUT_DIR=${base_dir}/bert_model_outputs - -mkdir -p $OUTPUT_DIR - -NCCL_TREE_THRESHOLD=0 deepspeed --launcher=mvapich ${base_dir}/../../deepspeed_train.py \ ---cf ${base_dir}/../../bert_large_lamb.json \ ---max_seq_length 128 \ ---output_dir $OUTPUT_DIR \ ---deepspeed_mpi \ ---deepspeed \ ---deepspeed_transformer_kernel \ ---print_steps 40 \ ---lr_schedule "EE" \ ---lr_offset 10e-4 \ ---job_name $JOB_NAME \ ---deepspeed_config ${base_dir}/deepspeed_bsz64k_onebitlamb_config_seq128_mpi_infiniband.json \ ---data_path_prefix /data/bert \ -&> ${JOB_NAME}.log diff --git a/training/bing_bert/1-bit_lamb/mpi_infiniband/mpi_train_bert_onebitlamb_bsz32k_seq512_infiniband.sh b/training/bing_bert/1-bit_lamb/mpi_infiniband/mpi_train_bert_onebitlamb_bsz32k_seq512_infiniband.sh deleted file mode 100644 index 101370ead..000000000 --- a/training/bing_bert/1-bit_lamb/mpi_infiniband/mpi_train_bert_onebitlamb_bsz32k_seq512_infiniband.sh +++ /dev/null @@ -1,50 +0,0 @@ -#!/bin/bash - -# If you are able to install pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs), -# we highly recommend you to use the NCCL-based 1-bit Lamb -# which has better performance and ease of use -# (see scripts in DeepSpeedExamples/bing_bert/1-bit_lamb/nccl -# and read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-lamb/) - -# Note: Please use the deepspeed launch script for most cases. -# For advanced users, mpirun or other MPI launchers can be used -# with this script as follows. -# mpirun -n 2 [launcher-args] python ${base_dir}/../deepspeed_train.py -# As an example, below we include the command we used - -base_dir=`pwd` - -# Assumes job name in previous seq128 run, will resume training from epoch 150 -EPOCH=150 - -# Where should we save checkpoints and tensorboard events? -JOB_NAME=onebit_lamb_32k_chkpt${EPOCH}_seq512_mpirun_infiniband -OUTPUT_DIR=${base_dir}/bert_model_outputs - -CHECKPOINT_BASE_PATH=${OUTPUT_DIR}/saved_models/onebit_lamb_64k_seq128_mpirun_infiniband -CHECKPOINT_NAME=`basename ${CHECKPOINT_BASE_PATH}/epoch${EPOCH}_*` -echo "checkpoint id: $CHECKPOINT_NAME" - -mkdir -p $OUTPUT_DIR - -mpirun -n 128 -ppn 8 -f /tmp/deepspeed_mvapich_hostfile -env MV2_SUPPORT_DL=1 -env MV2_USE_GDR=0 -env MV2_USE_CUDA=1 -env MV2_USE_GDRCOPY=0 -env MV2_SMP_USE_CMA=0 -env MV2_DEBUG_SHOW_BACKTRACE=1 python ${base_dir}/../../deepspeed_train.py \ ---cf ${base_dir}/../../bert_large_lamb.json \ ---max_seq_length 512 \ ---output_dir $OUTPUT_DIR \ ---print_steps 100 \ ---deepspeed_mpi \ ---deepspeed \ ---deepspeed_transformer_kernel \ ---job_name $JOB_NAME \ ---deepspeed_config ${base_dir}/deepspeed_bsz32k_onebitlamb_config_seq512_mpi_infiniband.json \ ---data_path_prefix /data/bert \ ---validation_data_path_prefix /data/bert \ ---rewarmup \ ---lr_schedule "EE" \ ---attention_dropout_checkpoint \ ---lr_offset 0.0 \ ---load_training_checkpoint ${CHECKPOINT_BASE_PATH} \ ---load_checkpoint_id ${CHECKPOINT_NAME} \ -&> ${JOB_NAME}.log diff --git a/training/bing_bert/1-bit_lamb/mpi_infiniband/mpi_train_bert_onebitlamb_bsz64k_seq128_infiniband.sh b/training/bing_bert/1-bit_lamb/mpi_infiniband/mpi_train_bert_onebitlamb_bsz64k_seq128_infiniband.sh deleted file mode 100644 index c376a4de2..000000000 --- a/training/bing_bert/1-bit_lamb/mpi_infiniband/mpi_train_bert_onebitlamb_bsz64k_seq128_infiniband.sh +++ /dev/null @@ -1,39 +0,0 @@ -#!/bin/bash - -# If you are able to install pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs), -# we highly recommend you to use the NCCL-based 1-bit Lamb -# which has better performance and ease of use -# (see scripts in DeepSpeedExamples/bing_bert/1-bit_lamb/nccl -# and read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-lamb/) - -# Note: Please use the deepspeed launch script for most cases. -# For advanced users, mpirun or other MPI launchers can be used -# with this script as follows. -# mpirun -n 2 [launcher-args] python ${base_dir}/../deepspeed_train.py -# As an example, below we include the command we used - -base_dir=`pwd` - -# Where should we save checkpoints and tensorboard events? -JOB_NAME=onebit_lamb_64k_seq128_mpirun_infiniband -OUTPUT_DIR=${base_dir}/bert_model_outputs - -mkdir -p $OUTPUT_DIR - -# LD_PRELOAD is used to load a specific nccl version. Remove it if needed. -mpirun -n 128 -ppn 8 -f /tmp/deepspeed_mvapich_hostfile -env MV2_SUPPORT_DL=1 -env MV2_USE_GDR=0 -env MV2_USE_CUDA=1 -env MV2_USE_GDRCOPY=0 -env MV2_SMP_USE_CMA=0 -env MV2_DEBUG_SHOW_BACKTRACE=1 python ${base_dir}/../../deepspeed_train.py \ ---cf ${base_dir}/../../bert_large_lamb.json \ ---max_seq_length 128 \ ---output_dir $OUTPUT_DIR \ ---deepspeed_mpi \ ---deepspeed \ ---deepspeed_transformer_kernel \ ---print_steps 40 \ ---lr_schedule "EE" \ ---lr_offset 10e-4 \ ---job_name $JOB_NAME \ ---deepspeed_config ${base_dir}/deepspeed_bsz64k_onebitlamb_config_seq128_mpi_infiniband.json \ ---data_path_prefix /data/bert \ -&> ${JOB_NAME}.log diff --git a/training/bing_bert/1-bit_lamb/nccl/deepspeed_bsz32k_onebitlamb_config_seq512_nccl.json b/training/bing_bert/1-bit_lamb/nccl/deepspeed_bsz32k_onebitlamb_config_seq512_nccl.json deleted file mode 100644 index 27670e0e3..000000000 --- a/training/bing_bert/1-bit_lamb/nccl/deepspeed_bsz32k_onebitlamb_config_seq512_nccl.json +++ /dev/null @@ -1,31 +0,0 @@ -{ - "train_batch_size": 32768, - "train_micro_batch_size_per_gpu": 16, - "steps_per_print": 1000, - "prescale_gradients": false, - "optimizer": { - "type": "OneBitLamb", - "params": { - "lr": 2e-3, - "weight_decay": 0.01, - "bias_correction": false, - "max_coeff": 0.3, - "min_coeff": 0.01, - "freeze_step": 6100, - "cuda_aware": false, - "comm_backend_name": "nccl", - "coeff_beta": 0.9, - "factor_max": 4.0, - "factor_min": 0.5, - "factor_threshold": 0.1 - } - }, - "gradient_clipping": 1.0, - - "wall_clock_breakdown": false, - - "fp16": { - "enabled": true, - "loss_scale": 0 - } -} diff --git a/training/bing_bert/1-bit_lamb/nccl/deepspeed_bsz64k_onebitlamb_config_seq128_nccl.json b/training/bing_bert/1-bit_lamb/nccl/deepspeed_bsz64k_onebitlamb_config_seq128_nccl.json deleted file mode 100644 index aa3dd599f..000000000 --- a/training/bing_bert/1-bit_lamb/nccl/deepspeed_bsz64k_onebitlamb_config_seq128_nccl.json +++ /dev/null @@ -1,32 +0,0 @@ -{ - "train_batch_size": 65536, - "train_micro_batch_size_per_gpu": 64, - "steps_per_print": 1000, - "prescale_gradients": false, - "optimizer": { - "type": "OneBitLamb", - "params": { - "lr": 11e-3, - "weight_decay": 0.01, - "bias_correction": false, - "max_coeff": 0.3, - "min_coeff": 0.01, - "freeze_step": 1000, - "cuda_aware": false, - "comm_backend_name": "nccl", - "coeff_beta": 0.9, - "factor_max": 4.0, - "factor_min": 0.5, - "factor_threshold": 0.1 - } - }, - "gradient_clipping": 1.0, - - "wall_clock_breakdown": false, - - "fp16": { - "enabled": true, - "loss_scale": 0, - "initial_scale_power": 16 - } -} diff --git a/training/bing_bert/1-bit_lamb/nccl/ds_train_bert_onebitlamb_bsz32k_seq512_nccl.sh b/training/bing_bert/1-bit_lamb/nccl/ds_train_bert_onebitlamb_bsz32k_seq512_nccl.sh deleted file mode 100644 index 0a0798b62..000000000 --- a/training/bing_bert/1-bit_lamb/nccl/ds_train_bert_onebitlamb_bsz32k_seq512_nccl.sh +++ /dev/null @@ -1,42 +0,0 @@ -#!/bin/bash - -# This script requires pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs). -# Read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-lamb - -base_dir=`pwd` - -# Assumes job name in previous seq128 run, will resume training from epoch 150 -EPOCH=150 - -# Where should we save checkpoints and tensorboard events? -JOB_NAME=onebit_lamb_32k_chkpt${EPOCH}_seq512_nccl -OUTPUT_DIR=${base_dir}/bert_model_outputs - -CHECKPOINT_BASE_PATH=${OUTPUT_DIR}/saved_models/onebit_lamb_64k_seq128_nccl -CHECKPOINT_NAME=`basename ${CHECKPOINT_BASE_PATH}/epoch${EPOCH}_*` -echo "checkpoint id: $CHECKPOINT_NAME" - -mkdir -p $OUTPUT_DIR - -# NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 are used to disable infiniband. Remove it if needed. -NCCL_TREE_THRESHOLD=0 NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 deepspeed ${base_dir}/../../deepspeed_train.py \ ---cf ${base_dir}/../../bert_large_lamb.json \ ---max_seq_length 512 \ ---output_dir $OUTPUT_DIR \ ---print_steps 100 \ ---deepspeed \ ---deepspeed_transformer_kernel \ ---job_name $JOB_NAME \ ---deepspeed_config ${base_dir}/deepspeed_bsz32k_onebitlamb_config_seq512_nccl.json \ ---data_path_prefix /data/bert \ ---validation_data_path_prefix /data/bert \ ---rewarmup \ ---lr_schedule "EE" \ ---attention_dropout_checkpoint \ ---lr_offset 0.0 \ ---load_training_checkpoint ${CHECKPOINT_BASE_PATH} \ ---load_checkpoint_id ${CHECKPOINT_NAME} \ ---ckpt_to_save 160 \ -&> ${JOB_NAME}.log diff --git a/training/bing_bert/1-bit_lamb/nccl/ds_train_bert_onebitlamb_bsz64k_seq128_nccl.sh b/training/bing_bert/1-bit_lamb/nccl/ds_train_bert_onebitlamb_bsz64k_seq128_nccl.sh deleted file mode 100644 index 315e7d596..000000000 --- a/training/bing_bert/1-bit_lamb/nccl/ds_train_bert_onebitlamb_bsz64k_seq128_nccl.sh +++ /dev/null @@ -1,30 +0,0 @@ -#!/bin/bash - -# This script requires pytorch >= 1.8 -# (and nccl >= 2.8.3 if you have 64 or more GPUs). -# Read the tutorial for more details: -# https://www.deepspeed.ai/tutorials/onebit-lamb - -base_dir=`pwd` - -# Where should we save checkpoints and tensorboard events? -JOB_NAME=onebit_lamb_64k_seq128_nccl -OUTPUT_DIR=${base_dir}/bert_model_outputs - -mkdir -p $OUTPUT_DIR - -# NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 are used to disable infiniband. Remove it if needed. -NCCL_TREE_THRESHOLD=0 NCCL_IB_DISABLE=1 NCCL_SOCKET_IFNAME=eth0 deepspeed ${base_dir}/../../deepspeed_train.py \ ---cf ${base_dir}/../../bert_large_lamb.json \ ---max_seq_length 128 \ ---output_dir $OUTPUT_DIR \ ---deepspeed \ ---deepspeed_transformer_kernel \ ---print_steps 40 \ ---lr_schedule "EE" \ ---lr_offset 10e-4 \ ---job_name $JOB_NAME \ ---deepspeed_config ${base_dir}/deepspeed_bsz64k_onebitlamb_config_seq128_nccl.json \ ---data_path_prefix /data/bert \ ---ckpt_to_save 150 \ -&> ${JOB_NAME}.log diff --git a/training/bing_bert/deepspeed_bsz4k_progressive_layer_drop_config_seq128.json b/training/bing_bert/deepspeed_bsz4k_progressive_layer_drop_config_seq128.json deleted file mode 100755 index 42f865cd7..000000000 --- a/training/bing_bert/deepspeed_bsz4k_progressive_layer_drop_config_seq128.json +++ /dev/null @@ -1,26 +0,0 @@ -{ - "train_batch_size": 4096, - "train_micro_batch_size_per_gpu": 16, - "steps_per_print": 1000, - "prescale_gradients": true, - "gradient_predivide_factor": 8, - "optimizer": { - "type": "Adam", - "params": { - "lr": 1e-3, - "weight_decay": 0.01, - "bias_correction": false - } - }, - "gradient_clipping": 1.0, - "wall_clock_breakdown": false, - "fp16": { - "enabled": true, - "loss_scale": 0 - }, - "progressive_layer_drop": { - "enabled": true, - "theta": 0.5, - "gamma": 0.001 - } -} diff --git a/training/bing_bert/deepspeed_bsz64k_lamb_config_seq128.json b/training/bing_bert/deepspeed_bsz64k_lamb_config_seq128.json index 2ae7ae5e4..d8f0457e2 100644 --- a/training/bing_bert/deepspeed_bsz64k_lamb_config_seq128.json +++ b/training/bing_bert/deepspeed_bsz64k_lamb_config_seq128.json @@ -20,15 +20,5 @@ "fp16": { "enabled": true, "loss_scale": 0 - }, - "sparse_attention": { - "mode": "fixed", - "block": 16, - "different_layout_per_head": true, - "num_local_blocks": 4, - "num_global_blocks": 1, - "attention": "bidirectional", - "horizontal_global_attention": false, - "num_different_global_patterns": 4 } } diff --git a/training/bing_bert/deepspeed_train.py b/training/bing_bert/deepspeed_train.py index 4e3bc4abc..6626a307a 100755 --- a/training/bing_bert/deepspeed_train.py +++ b/training/bing_bert/deepspeed_train.py @@ -365,8 +365,6 @@ def construct_arguments(): def prepare_optimizer_parameters(args, model): config = args.config - deepspeed_config = json.load( - open(args.deepspeed_config, 'r', encoding='utf-8')) param_optimizer = list(model.network.named_parameters()) param_optimizer = [n for n in param_optimizer if 'pooler' not in n[0]] @@ -381,74 +379,19 @@ def prepare_optimizer_parameters(args, model): else: weight_decay = 0.01 - if deepspeed_config["optimizer"]["type"] not in [ - "OneBitAdam", "OneBitLamb", "ZeroOneAdam" - ]: - optimizer_grouped_parameters = [{ - 'params': [ - p for n, p in param_optimizer - if not any(nd in n for nd in no_decay) - ], - 'weight_decay': - weight_decay - }, { - 'params': - [p for n, p in param_optimizer if any(nd in n for nd in no_decay)], - 'weight_decay': - 0.0 - }] - else: - # Because 1-bit compression cannot represent exact zero, it is required to - # provide a momentum mask for those params that have constant exact zeros in their - # momentums, otherwise the compression error would keep accumulating. - # For example, for bert pre-training seq 128, bert.embeddings.position_embeddings.weight - # always have exact zeros in its momentum for row 129 to 512, because it only - # learns up to seq length 128 while the model supports up to 512 seq length. - need_mask = ['position_embeddings.weight'] - need_mask_p = [] - need_mask_decay = [] - masks = [] - for n, p in param_optimizer: - if any(nd in n for nd in need_mask): - mask = torch.zeros_like(p.data) - for position in range(args.max_seq_length): - for col in range(p.size()[1]): - mask[position][col] += 1 - if deepspeed_config["optimizer"]["type"] in ["OneBitAdam", "ZeroOneAdam"]: - mask = torch.flatten(mask) - masks.append(mask) - need_mask_p.append(p) - if any(nd in n for nd in no_decay): - need_mask_decay.append(0.0) - else: - need_mask_decay.append(weight_decay) - - optimizer_grouped_parameters = [{ - 'params': [ - p for n, p in param_optimizer - if not any(nd in n for nd in no_decay + need_mask) - ], - 'weight_decay': - weight_decay - }, { - 'params': [ - p for n, p in param_optimizer - if (any(nd in n - for nd in no_decay) and not any(nd in n - for nd in need_mask)) - ], - 'weight_decay': - 0.0 - }] - - for i_mask in range(len(need_mask_p)): - optimizer_grouped_parameters.append({ - 'params': [need_mask_p[i_mask]], - 'weight_decay': - need_mask_decay[i_mask], - 'exp_avg_mask': - masks[i_mask] - }) + optimizer_grouped_parameters = [{ + 'params': [ + p for n, p in param_optimizer + if not any(nd in n for nd in no_decay) + ], + 'weight_decay': + weight_decay + }, { + 'params': + [p for n, p in param_optimizer if any(nd in n for nd in no_decay)], + 'weight_decay': + 0.0 + }] return optimizer_grouped_parameters diff --git a/training/bing_bert/ds_sa_train_bert_bsz64k_seq128.sh b/training/bing_bert/ds_sa_train_bert_bsz64k_seq128.sh deleted file mode 100644 index 3132cc2b8..000000000 --- a/training/bing_bert/ds_sa_train_bert_bsz64k_seq128.sh +++ /dev/null @@ -1,25 +0,0 @@ -#!/bin/bash - -# This script runs deepspeed using sparse attention for BertEncoderLayer. - -base_dir=`pwd` - -# Where should we save checkpoints and tensorboard events? -JOB_NAME=lamb_64k_seq128 -OUTPUT_DIR=${base_dir}/bert_model_outputs - -mkdir -p $OUTPUT_DIR - -NCCL_TREE_THRESHOLD=0 deepspeed ${base_dir}/deepspeed_train.py \ ---cf ${base_dir}/bert_large_lamb.json \ ---max_seq_length 128 \ ---output_dir $OUTPUT_DIR \ ---deepspeed \ ---deepspeed_sparse_attention \ ---print_steps 100 \ ---lr_schedule "EE" \ ---lr_offset 10e-4 \ ---job_name $JOB_NAME \ ---deepspeed_config ${base_dir}/deepspeed_bsz64k_lamb_config_seq128.json \ ---data_path_prefix /data/bert \ -&> ${JOB_NAME}.log diff --git a/training/bing_bert/ds_train_bert_progressive_layer_drop_bsz4k_seq128.sh b/training/bing_bert/ds_train_bert_progressive_layer_drop_bsz4k_seq128.sh deleted file mode 100755 index e0bc783dd..000000000 --- a/training/bing_bert/ds_train_bert_progressive_layer_drop_bsz4k_seq128.sh +++ /dev/null @@ -1,25 +0,0 @@ -#!/bin/bash - -base_dir=`pwd` - -# Where should we save checkpoints and tensorboard events? -JOB_NAME=adam_4k_seq128_progressive_layer_drop -OUTPUT_DIR=${base_dir}/bert_model_outputs - -mkdir -p $OUTPUT_DIR - -config="--progressive_layer_drop" - -NCCL_TREE_THRESHOLD=0 deepspeed \ -${base_dir}/deepspeed_train.py \ ---cf ${base_dir}/bert_base_large_lr.json \ ---max_seq_length 128 \ ---output_dir $OUTPUT_DIR \ ---deepspeed \ ---print_steps 100 \ ---lr_schedule "LE" \ ---job_name $JOB_NAME \ ---deepspeed_config ${base_dir}/deepspeed_bsz4k_progressive_layer_drop_config_seq128.json \ ---data_path_prefix /data/bert \ -${config} \ -&> ${JOB_NAME}.log diff --git a/training/bing_bert/nvidia/modelingpreln.py b/training/bing_bert/nvidia/modelingpreln.py index 9856f0607..87dc7f69a 100755 --- a/training/bing_bert/nvidia/modelingpreln.py +++ b/training/bing_bert/nvidia/modelingpreln.py @@ -75,44 +75,6 @@ def get_deepspeed_config(args): raise RuntimeError('deepspeed_config is not found in args.') -def get_sparse_attention_config(args, num_heads): - if args.deepspeed_sparse_attention: - ds_config = get_deepspeed_config(args) - if hasattr(ds_config, - 'sparse_attention') and ds_config.sparse_attention: - sa_config = ds_config.sparse_attention - sa_mode = sa_config.get('mode') - if (sa_mode == 'dense'): - from deepspeed.ops.sparse_attention import DenseSparsityConfig as STConfig - elif (sa_mode == 'fixed'): - from deepspeed.ops.sparse_attention import FixedSparsityConfig as STConfig - elif (sa_mode == 'bigbird'): - from deepspeed.ops.sparse_attention import BigBirdSparsityConfig as STConfig - elif (sa_mode == 'bslongformer'): - from deepspeed.ops.sparse_attention import BSLongformerSparsityConfig as STConfig - elif (sa_mode == 'variable'): - from deepspeed.ops.sparse_attention import VariableSparsityConfig as STConfig - else: - raise NotImplementedError( - f'Given sparsity mode, {sa_mode}, has not been implemented yet!' - ) - del sa_config['mode'] - return STConfig(num_heads=num_heads, **sa_config) - else: - from deepspeed.ops.sparse_attention import FixedSparsityConfig as STConfig - print( - 'deepspeed sparse attention is not set; Fixed sparsity is used as default.' - ) - return STConfig(num_heads=num_heads) - else: - return None - -def get_sparse_attention_utils(sparse_attention_config): - if sparse_attention_config is not None: - from deepspeed.ops.sparse_attention import SparseAttentionUtils - return SparseAttentionUtils - return None - def load_tf_weights_in_bert(model, tf_checkpoint_path): """ Load tf checkpoints in a pytorch model """ @@ -557,17 +519,12 @@ def forward(self, hidden_states, attention_mask): class BertEncoder(nn.Module): - def __init__(self, config, args, sparse_attention_config=None): + def __init__(self, config, args): super(BertEncoder, self).__init__() #Added later to make it similar to GPT-2 self.FinalLayerNorm = BertLayerNorm(config.hidden_size, eps=1e-12) - if args.deepspeed_transformer_kernel and args.deepspeed_sparse_attention: - raise NotImplementedError( - f'Currently DeepSpeed Transformer Kernels do not support Sparse Attention. To use Sparse Attention, you need to disable Transformer Kernels!' - ) - if args.deepspeed_transformer_kernel: from deepspeed import DeepSpeedTransformerLayer, DeepSpeedTransformerConfig @@ -597,12 +554,6 @@ def __init__(self, config, args, sparse_attention_config=None): ]) else: layer = BertLayer(config) - if sparse_attention_config is not None: - from deepspeed.ops.sparse_attention import BertSparseSelfAttention - - layer.attention.self = BertSparseSelfAttention( - config, sparsity_config=sparse_attention_config) - self.layer = nn.ModuleList([ copy.deepcopy(layer) for _ in range(config.num_hidden_layers) ]) @@ -991,15 +942,7 @@ class BertModel(BertPreTrainedModel): def __init__(self, config, args=None): super(BertModel, self).__init__(config) self.embeddings = BertEmbeddings(config) - # set pad_token_id that is used for sparse attention padding - self.pad_token_id = config.pad_token_id if hasattr( - config, 'pad_token_id') and config.pad_token_id is not None else 0 - # set sparse_attention_config if it has been selected - self.sparse_attention_config = get_sparse_attention_config( - args, config.num_attention_heads) - self.sparse_attention_utils = get_sparse_attention_utils(self.sparse_attention_config) - self.encoder = BertEncoder( - config, args, sparse_attention_config=self.sparse_attention_config) + self.encoder = BertEncoder(config, args) self.pooler = BertPooler(config) self.apply(self.init_bert_weights) logger.info("Init BERT pretrain model") @@ -1031,18 +974,6 @@ def forward(self, dtype=next(self.parameters()).dtype) # fp16 compatibility extended_attention_mask = (1.0 - extended_attention_mask) * -10000.0 - # If BertEncoder uses sparse attention, it needs to be padded based on the sparse attention block size - if self.sparse_attention_config is not None: - pad_len, input_ids, attention_mask, token_type_ids, position_ids, inputs_embeds = self.sparse_attention_utils.pad_to_block_size( - block_size=self.sparse_attention_config.block, - input_ids=input_ids, - attention_mask=extended_attention_mask, - token_type_ids=token_type_ids, - position_ids=None, - inputs_embeds=None, - pad_token_id=self.pad_token_id, - model_embeddings=self.embeddings) - embedding_output = self.embeddings(input_ids, token_type_ids) encoded_layers = self.encoder( embedding_output, @@ -1052,11 +983,6 @@ def forward(self, sequence_output = encoded_layers[-1] pooled_output = self.pooler(sequence_output) - # If BertEncoder uses sparse attention, and input_ids were padded, sequence output needs to be unpadded to original length - if self.sparse_attention_config is not None and pad_len > 0: - encoded_layers[-1] = self.sparse_attention_utils.unpad_sequence_output( - pad_len, encoded_layers[-1]) - if not output_all_encoded_layers: encoded_layers = encoded_layers[-1] return encoded_layers, pooled_output diff --git a/training/bing_bert/nvidia/modelingpreln_layerdrop.py b/training/bing_bert/nvidia/modelingpreln_layerdrop.py deleted file mode 100755 index b5beb89af..000000000 --- a/training/bing_bert/nvidia/modelingpreln_layerdrop.py +++ /dev/null @@ -1,1662 +0,0 @@ -# DeepSpeed note, code taken from commit 3d59216cec89a363649b4fe3d15295ba936ced0f -# https://github.com/NVIDIA/DeepLearningExamples/blob/master/PyTorch/LanguageModeling/BERT/modeling.py - -# coding=utf-8 -# Copyright 2018 The Google AI Language Team Authors and The HugginFace Inc. team. -# Copyright (c) 2018, NVIDIA CORPORATION. All rights reserved. -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. -"""PyTorch BERT model.""" - -from __future__ import absolute_import, division, print_function, unicode_literals - -import copy -import json -import logging -import math -import os -import shutil -import tarfile -import tempfile -import sys -from io import open - -import torch -from torch import nn -from torch.nn import CrossEntropyLoss -from torch.utils import checkpoint - -from turing.file_utils import cached_path - -from torch.nn import Module -from torch.nn.parameter import Parameter -import torch.nn.functional as F -import torch.nn.init as init - -import numpy as np - -logger = logging.getLogger(__name__) - -PRETRAINED_MODEL_ARCHIVE_MAP = { - 'bert-base-uncased': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-base-uncased.tar.gz", - 'bert-large-uncased': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-large-uncased.tar.gz", - 'bert-base-cased': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-base-cased.tar.gz", - 'bert-large-cased': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-large-cased.tar.gz", - 'bert-base-multilingual-uncased': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-base-multilingual-uncased.tar.gz", - 'bert-base-multilingual-cased': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-base-multilingual-cased.tar.gz", - 'bert-base-chinese': - "https://s3.amazonaws.com/models.huggingface.co/bert/bert-base-chinese.tar.gz", -} -CONFIG_NAME = 'bert_config.json' -WEIGHTS_NAME = 'pytorch_model.bin' -TF_WEIGHTS_NAME = 'model.ckpt' - - -def get_deepspeed_config(args): - if hasattr(args, 'deepspeed_config') and args.deepspeed_config: - from deepspeed import DeepSpeedConfig - return DeepSpeedConfig(args.deepspeed_config) - else: - raise RuntimeError('deepspeed_config is not found in args.') - - -def get_sparse_attention_config(args, num_heads): - if args.deepspeed_sparse_attention: - ds_config = get_deepspeed_config(args) - if hasattr(ds_config, - 'sparse_attention') and ds_config.sparse_attention: - sa_config = ds_config.sparse_attention - sa_mode = sa_config.get('mode') - if (sa_mode == 'dense'): - from deepspeed.ops.sparse_attention import DenseSparsityConfig as STConfig - elif (sa_mode == 'fixed'): - from deepspeed.ops.sparse_attention import FixedSparsityConfig as STConfig - elif (sa_mode == 'bigbird'): - from deepspeed.ops.sparse_attention import BigBirdSparsityConfig as STConfig - elif (sa_mode == 'bslongformer'): - from deepspeed.ops.sparse_attention import BSLongformerSparsityConfig as STConfig - elif (sa_mode == 'variable'): - from deepspeed.ops.sparse_attention import VariableSparsityConfig as STConfig - else: - raise NotImplementedError( - f'Given sparsity mode, {sa_mode}, has not been implemented yet!' - ) - del sa_config['mode'] - return STConfig(num_heads=num_heads, **sa_config) - else: - from deepspeed.ops.sparse_attention import FixedSparsityConfig as STConfig - print( - 'deepspeed sparse attention is not set; Fixed sparsity is used as default.' - ) - return STConfig(num_heads=num_heads) - else: - return None - - -def get_sparse_attention_utils(sparse_attention_config): - if sparse_attention_config is not None: - from deepspeed.ops.sparse_attention import SparseAttentionUtils - return SparseAttentionUtils - return None - - -def load_tf_weights_in_bert(model, tf_checkpoint_path): - """ Load tf checkpoints in a pytorch model - """ - try: - import re - import numpy as np - import tensorflow as tf - except ImportError: - print( - "Loading a TensorFlow models in PyTorch, requires TensorFlow to be installed. Please see " - "https://www.tensorflow.org/install/ for installation instructions." - ) - raise - tf_path = os.path.abspath(tf_checkpoint_path) - print("Converting TensorFlow checkpoint from {}".format(tf_path)) - # Load weights from TF model - init_vars = tf.train.list_variables(tf_path) - names = [] - arrays = [] - for name, shape in init_vars: - print("Loading TF weight {} with shape {}".format(name, shape)) - array = tf.train.load_variable(tf_path, name) - names.append(name) - arrays.append(array) - - for name, array in zip(names, arrays): - name = name.split('/') - # adam_v and adam_m are variables used in AdamWeightDecayOptimizer to calculated m and v - # which are not required for using pretrained model - if any(n in ["adam_v", "adam_m"] for n in name): - print("Skipping {}".format("/".join(name))) - continue - pointer = model - for m_name in name: - if re.fullmatch(r'[A-Za-z]+_\d+', m_name): - l = re.split(r'_(\d+)', m_name) - else: - l = [m_name] - if l[0] == 'kernel' or l[0] == 'gamma': - pointer = getattr(pointer, 'weight') - elif l[0] == 'output_bias' or l[0] == 'beta': - pointer = getattr(pointer, 'bias') - elif l[0] == 'output_weights': - pointer = getattr(pointer, 'weight') - else: - pointer = getattr(pointer, l[0]) - if len(l) >= 2: - num = int(l[1]) - pointer = pointer[num] - if m_name[-11:] == '_embeddings': - pointer = getattr(pointer, 'weight') - elif m_name == 'kernel': - array = np.transpose(array) - try: - assert pointer.shape == array.shape - except AssertionError as e: - e.args += (pointer.shape, array.shape) - raise - print("Initialize PyTorch weight {}".format(name)) - pointer.data = torch.from_numpy(array) - return model - - -@torch.jit.script -def f_gelu(x): - pdtype = x.dtype - x = x.float() - y = x * 0.5 * (1.0 + torch.erf(x / math.sqrt(2.0))) - return y.to(pdtype) - - -@torch.jit.script -def bias_gelu(bias, y): - x = bias + y - return x * 0.5 * (1.0 + torch.erf(x / 1.41421)) - - -@torch.jit.script -def bias_tanh(bias, y): - x = bias + y - return torch.tanh(x) - - -def gelu(x): - """Implementation of the gelu activation function. - For information: OpenAI GPT's gelu is slightly different (and gives slightly different results): - 0.5 * x * (1 + torch.tanh(math.sqrt(2 / math.pi) * (x + 0.044715 * torch.pow(x, 3)))) - Also see https://arxiv.org/abs/1606.08415 - """ - return f_gelu(x) - - -def swish(x): - return x * torch.sigmoid(x) - - -ACT2FN = {"gelu": gelu, "relu": torch.nn.functional.relu, "swish": swish} - - -class LinearActivation(Module): - r"""Fused Linear and activation Module. - """ - __constants__ = ['bias'] - - def __init__(self, in_features, out_features, act='gelu', bias=True): - super(LinearActivation, self).__init__() - self.in_features = in_features - self.out_features = out_features - self.fused_gelu = False - self.fused_tanh = False - if isinstance(act, str) or (sys.version_info[0] == 2 - and isinstance(act, unicode)): - if bias and act == 'gelu': - self.fused_gelu = True - elif bias and act == 'tanh': - self.fused_tanh = True - else: - self.act_fn = ACT2FN[act] - else: - self.act_fn = act - self.weight = Parameter(torch.Tensor(out_features, in_features)) - if bias: - self.bias = Parameter(torch.Tensor(out_features)) - else: - self.register_parameter('bias', None) - self.reset_parameters() - - def reset_parameters(self): - init.kaiming_uniform_(self.weight, a=math.sqrt(5)) - if self.bias is not None: - fan_in, _ = init._calculate_fan_in_and_fan_out(self.weight) - bound = 1 / math.sqrt(fan_in) - init.uniform_(self.bias, -bound, bound) - - def forward(self, input): - if self.fused_gelu: - return bias_gelu(self.bias, F.linear(input, self.weight, None)) - elif self.fused_tanh: - return bias_tanh(self.bias, F.linear(input, self.weight, None)) - else: - return self.act_fn(F.linear(input, self.weight, self.bias)) - - def extra_repr(self): - return 'in_features={}, out_features={}, bias={}'.format( - self.in_features, self.out_features, self.bias is not None) - - -class BertConfig(object): - """Configuration class to store the configuration of a `BertModel`. - """ - def __init__(self, - vocab_size_or_config_json_file, - hidden_size=768, - num_hidden_layers=12, - num_attention_heads=12, - intermediate_size=3072, - hidden_act="gelu", - hidden_dropout_prob=0.1, - attention_probs_dropout_prob=0.1, - max_position_embeddings=512, - type_vocab_size=2, - initializer_range=0.02): - """Constructs BertConfig. - - Args: - vocab_size_or_config_json_file: Vocabulary size of `inputs_ids` in `BertModel`. - hidden_size: Size of the encoder layers and the pooler layer. - num_hidden_layers: Number of hidden layers in the Transformer encoder. - num_attention_heads: Number of attention heads for each attention layer in - the Transformer encoder. - intermediate_size: The size of the "intermediate" (i.e., feed-forward) - layer in the Transformer encoder. - hidden_act: The non-linear activation function (function or string) in the - encoder and pooler. If string, "gelu", "relu" and "swish" are supported. - hidden_dropout_prob: The dropout probabilitiy for all fully connected - layers in the embeddings, encoder, and pooler. - attention_probs_dropout_prob: The dropout ratio for the attention - probabilities. - max_position_embeddings: The maximum sequence length that this model might - ever be used with. Typically set this to something large just in case - (e.g., 512 or 1024 or 2048). - type_vocab_size: The vocabulary size of the `token_type_ids` passed into - `BertModel`. - initializer_range: The sttdev of the truncated_normal_initializer for - initializing all weight matrices. - """ - if isinstance(vocab_size_or_config_json_file, - str) or (sys.version_info[0] == 2 and isinstance( - vocab_size_or_config_json_file, unicode)): - with open(vocab_size_or_config_json_file, "r", - encoding='utf-8') as reader: - json_config = json.loads(reader.read()) - for key, value in json_config.items(): - self.__dict__[key] = value - elif isinstance(vocab_size_or_config_json_file, int): - self.vocab_size = vocab_size_or_config_json_file - self.hidden_size = hidden_size - self.num_hidden_layers = num_hidden_layers - self.num_attention_heads = num_attention_heads - self.hidden_act = hidden_act - self.intermediate_size = intermediate_size - self.hidden_dropout_prob = hidden_dropout_prob - self.attention_probs_dropout_prob = attention_probs_dropout_prob - self.max_position_embeddings = max_position_embeddings - self.type_vocab_size = type_vocab_size - self.initializer_range = initializer_range - else: - raise ValueError( - "First argument must be either a vocabulary size (int)" - "or the path to a pretrained model config file (str)") - - @classmethod - def from_dict(cls, json_object): - """Constructs a `BertConfig` from a Python dictionary of parameters.""" - config = BertConfig(vocab_size_or_config_json_file=-1) - for key, value in json_object.items(): - config.__dict__[key] = value - return config - - @classmethod - def from_json_file(cls, json_file): - """Constructs a `BertConfig` from a json file of parameters.""" - with open(json_file, "r", encoding='utf-8') as reader: - text = reader.read() - return cls.from_dict(json.loads(text)) - - def __repr__(self): - return str(self.to_json_string()) - - def to_dict(self): - """Serializes this instance to a Python dictionary.""" - output = copy.deepcopy(self.__dict__) - return output - - def to_json_string(self): - """Serializes this instance to a JSON string.""" - return json.dumps(self.to_dict(), indent=2, sort_keys=True) + "\n" - - -try: - import apex - #apex.amp.register_half_function(apex.normalization.fused_layer_norm, 'FusedLayerNorm') - import apex.normalization - #apex.amp.register_float_function(apex.normalization.FusedLayerNorm, 'forward') - BertLayerNorm = apex.normalization.FusedLayerNorm -except ImportError: - print( - "Better speed can be achieved with apex installed from https://www.github.com/nvidia/apex." - ) - - class BertLayerNorm(nn.Module): - def __init__(self, hidden_size, eps=1e-12): - """Construct a layernorm module in the TF style (epsilon inside the square root). - """ - super(BertLayerNorm, self).__init__() - self.weight = nn.Parameter(torch.ones(hidden_size)) - self.bias = nn.Parameter(torch.zeros(hidden_size)) - self.variance_epsilon = eps - - def forward(self, x): - pdtype = x.dtype - x = x.float() - u = x.mean(-1, keepdim=True) - s = (x - u).pow(2).mean(-1, keepdim=True) - x = (x - u) / torch.sqrt(s + self.variance_epsilon) - return self.weight * x.to(pdtype) + self.bias - - -class BertEmbeddings(nn.Module): - """Construct the embeddings from word, position and token_type embeddings. - """ - def __init__(self, config): - super(BertEmbeddings, self).__init__() - self.word_embeddings = nn.Embedding(config.vocab_size, - config.hidden_size) - self.position_embeddings = nn.Embedding(config.max_position_embeddings, - config.hidden_size) - self.token_type_embeddings = nn.Embedding(config.type_vocab_size, - config.hidden_size) - - # self.LayerNorm is not snake-cased to stick with TensorFlow model variable name and be able to load - # any TensorFlow checkpoint file - self.LayerNorm = BertLayerNorm(config.hidden_size, eps=1e-12) - self.dropout = nn.Dropout(config.hidden_dropout_prob) - - def forward(self, input_ids, token_type_ids=None): - seq_length = input_ids.size(1) - position_ids = torch.arange(seq_length, - dtype=torch.long, - device=input_ids.device) - position_ids = position_ids.unsqueeze(0).expand_as(input_ids) - if token_type_ids is None: - token_type_ids = torch.zeros_like(input_ids) - - words_embeddings = self.word_embeddings(input_ids) - position_embeddings = self.position_embeddings(position_ids) - token_type_embeddings = self.token_type_embeddings(token_type_ids) - - embeddings = words_embeddings + position_embeddings + token_type_embeddings - embeddings = self.LayerNorm(embeddings) - embeddings = self.dropout(embeddings) - return embeddings - - -class BertSelfAttention(nn.Module): - def __init__(self, config): - super(BertSelfAttention, self).__init__() - if config.hidden_size % config.num_attention_heads != 0: - raise ValueError( - "The hidden size (%d) is not a multiple of the number of attention " - "heads (%d)" % - (config.hidden_size, config.num_attention_heads)) - self.num_attention_heads = config.num_attention_heads - self.attention_head_size = int(config.hidden_size / - config.num_attention_heads) - self.all_head_size = self.num_attention_heads * self.attention_head_size - - self.query = nn.Linear(config.hidden_size, self.all_head_size) - self.key = nn.Linear(config.hidden_size, self.all_head_size) - self.value = nn.Linear(config.hidden_size, self.all_head_size) - - self.dropout = nn.Dropout(config.attention_probs_dropout_prob) - self.softmax = nn.Softmax(dim=-1) - - def transpose_for_scores(self, x): - new_x_shape = x.size()[:-1] + (self.num_attention_heads, - self.attention_head_size) - x = x.view(*new_x_shape) - return x.permute(0, 2, 1, 3) - - def transpose_key_for_scores(self, x): - new_x_shape = x.size()[:-1] + (self.num_attention_heads, - self.attention_head_size) - x = x.view(*new_x_shape) - return x.permute(0, 2, 3, 1) - - def forward(self, hidden_states, attention_mask): - mixed_query_layer = self.query(hidden_states) - mixed_key_layer = self.key(hidden_states) - mixed_value_layer = self.value(hidden_states) - - query_layer = self.transpose_for_scores(mixed_query_layer) - key_layer = self.transpose_key_for_scores(mixed_key_layer) - value_layer = self.transpose_for_scores(mixed_value_layer) - - # Take the dot product between "query" and "key" to get the raw attention scores. - attention_scores = torch.matmul(query_layer, key_layer) - attention_scores = attention_scores / math.sqrt( - self.attention_head_size) - # Apply the attention mask is (precomputed for all layers in BertModel forward() function) - attention_scores = attention_scores + attention_mask - - pdtype = attention_scores.dtype - # Normalize the attention scores to probabilities. - attention_probs = self.softmax(attention_scores) - - # This is actually dropping out entire tokens to attend to, which might - # seem a bit unusual, but is taken from the original Transformer paper. - attention_probs = self.dropout(attention_probs) - - context_layer = torch.matmul(attention_probs, value_layer) - context_layer = context_layer.permute(0, 2, 1, 3).contiguous() - new_context_layer_shape = context_layer.size()[:-2] + ( - self.all_head_size, ) - context_layer = context_layer.view(*new_context_layer_shape) - return context_layer - - -class BertSelfOutput(nn.Module): - def __init__(self, config): - super(BertSelfOutput, self).__init__() - self.dense = nn.Linear(config.hidden_size, config.hidden_size) - self.dense.bert_output_layer = True - self.dropout = nn.Dropout(config.hidden_dropout_prob) - - def forward(self, hidden_states, input_tensor): - hidden_states = self.dense(hidden_states) - hidden_states = self.dropout(hidden_states) - return hidden_states - - -class BertAttention(nn.Module): - def __init__(self, config): - super(BertAttention, self).__init__() - self.self = BertSelfAttention(config) - self.output = BertSelfOutput(config) - - def forward(self, input_tensor, attention_mask): - self_output = self.self(input_tensor, attention_mask) - attention_output = self.output(self_output, input_tensor) - return attention_output - - -class BertIntermediate(nn.Module): - def __init__(self, config): - super(BertIntermediate, self).__init__() - self.dense_act = LinearActivation(config.hidden_size, - config.intermediate_size, - act=config.hidden_act) - - def forward(self, hidden_states): - hidden_states = self.dense_act(hidden_states) - return hidden_states - - -class BertOutput(nn.Module): - def __init__(self, config): - super(BertOutput, self).__init__() - self.dense = nn.Linear(config.intermediate_size, config.hidden_size) - self.dense.bert_output_layer = True - self.dropout = nn.Dropout(config.hidden_dropout_prob) - - def forward(self, hidden_states): - hidden_states = self.dense(hidden_states) - hidden_states = self.dropout(hidden_states) - return hidden_states - - -class BertLayer(nn.Module): - def __init__(self, config): - super(BertLayer, self).__init__() - self.attention = BertAttention(config) - self.PreAttentionLayerNorm = BertLayerNorm(config.hidden_size, - eps=1e-12) - self.PostAttentionLayerNorm = BertLayerNorm(config.hidden_size, - eps=1e-12) - self.intermediate = BertIntermediate(config) - self.output = BertOutput(config) - - def forward(self, hidden_states, attention_mask, action=1, keep_prob=1.0): - if action == 0: - intermediate_input = hidden_states - else: - input_layer_norm = self.PreAttentionLayerNorm(hidden_states) - attention_output = self.attention(input_layer_norm, attention_mask) - attention_output = attention_output * 1 / keep_prob - intermediate_input = hidden_states + attention_output - - if action == 0: - layer_output = intermediate_input - else: - intermediate_layer_norm = self.PostAttentionLayerNorm( - intermediate_input) - intermediate_output = self.intermediate(intermediate_layer_norm) - layer_output = self.output(intermediate_output) - layer_output = layer_output * 1 / keep_prob - layer_output = layer_output + intermediate_input - - return layer_output - - -class BertEncoder(nn.Module): - def __init__(self, config, args, sparse_attention_config=None): - super(BertEncoder, self).__init__() - - #Added later to make it similar to GPT-2 - self.FinalLayerNorm = BertLayerNorm(config.hidden_size, eps=1e-12) - - if args.deepspeed_transformer_kernel and args.deepspeed_sparse_attention: - raise NotImplementedError( - f'Currently DeepSpeed Transformer Kernels do not support Sparse Attention. To use Sparse Attention, you need to disable Transformer Kernels!' - ) - - if args.deepspeed_transformer_kernel: - from deepspeed import DeepSpeedTransformerLayer, DeepSpeedTransformerConfig - - ds_config = get_deepspeed_config(args) - cuda_config = DeepSpeedTransformerConfig( - batch_size=ds_config.train_micro_batch_size_per_gpu, - max_seq_length=args.max_seq_length, - hidden_size=config.hidden_size, - intermediate_size=config.intermediate_size, - heads=config.num_attention_heads, - attn_dropout_ratio=config.attention_probs_dropout_prob, - hidden_dropout_ratio=config.hidden_dropout_prob, - num_hidden_layers=config.num_hidden_layers, - initializer_range=config.initializer_range, - local_rank=args.local_rank - if hasattr(args, 'local_rank') else -1, - seed=args.seed, - fp16=ds_config.fp16_enabled, - pre_layer_norm=True, - attn_dropout_checkpoint=args.attention_dropout_checkpoint, - normalize_invertible=args.normalize_invertible, - gelu_checkpoint=args.gelu_checkpoint, - stochastic_mode=args.stochastic_mode) - - self.layer = nn.ModuleList([ - copy.deepcopy(DeepSpeedTransformerLayer(i, cuda_config)) - for i in range(config.num_hidden_layers) - ]) - else: - layer = BertLayer(config) - if sparse_attention_config is not None: - from deepspeed.ops.sparse_attention import BertSparseSelfAttention - - layer.attention.self = BertSparseSelfAttention( - config, sparsity_config=sparse_attention_config) - - self.layer = nn.ModuleList([ - copy.deepcopy(layer) for _ in range(config.num_hidden_layers) - ]) - - # def forward(self, hidden_states, attention_mask, output_all_encoded_layers=True): - # all_encoder_layers = [] - # for layer_module in self.layer: - # hidden_states = layer_module(hidden_states, attention_mask) - # if output_all_encoded_layers: - # all_encoder_layers.append(hidden_states) - # if not output_all_encoded_layers: - # all_encoder_layers.append(hidden_states) - # return all_encoder_layers - def forward(self, - hidden_states, - attention_mask, - output_all_encoded_layers=True, - checkpoint_activations=False, - progressive_layer_drop=False, - theta=0.5): - all_encoder_layers = [] - - def custom(start, end): - def custom_forward(*inputs): - layers = self.layer[start:end] - x_ = inputs[0] - for layer in layers: - x_ = layer(x_, inputs[1]) - return x_ - - return custom_forward - - if checkpoint_activations: - l = 0 - num_layers = len(self.layer) - chunk_length = math.ceil(math.sqrt(num_layers)) - while l < num_layers: - hidden_states = checkpoint.checkpoint( - custom(l, l + chunk_length), hidden_states, - attention_mask * 1) - l += chunk_length - # decoder layers - else: - if not progressive_layer_drop: - for i, layer_module in enumerate(self.layer): - hidden_states = layer_module(hidden_states, attention_mask) - - if output_all_encoded_layers: - all_encoder_layers.append(hidden_states) - else: - drop_prob = 1 - theta - step = drop_prob / len(self.layer) - p = 1.0 - # print("+ stochastic drop, depth, Theta {}:".format(theta)) - - for i, layer_module in enumerate(self.layer): - - action = np.random.choice([1, 0], p=[p, 1 - p]) - p = p - step - hidden_states = layer_module(hidden_states, attention_mask, - action, p) - if output_all_encoded_layers: - all_encoder_layers.append(hidden_states) - - if not output_all_encoded_layers or checkpoint_activations: - hidden_states = self.FinalLayerNorm(hidden_states) - all_encoder_layers.append(hidden_states) - return all_encoder_layers - - -#class BertEncoder(nn.Module): -# def __init__(self, config): -# super(BertEncoder, self).__init__() -# layer = BertLayer(config) -# self.layer = nn.ModuleList([copy.deepcopy(layer) for _ in range(config.num_hidden_layers)]) -# -# def forward(self, hidden_states, attention_mask, output_all_encoded_layers=True): -# all_encoder_layers = [] -# for layer_module in self.layer: -# hidden_states = layer_module(hidden_states, attention_mask) -# if output_all_encoded_layers: -# all_encoder_layers.append(hidden_states) -# if not output_all_encoded_layers: -# all_encoder_layers.append(hidden_states) -# return all_encoder_layers - - -class BertPooler(nn.Module): - def __init__(self, config): - super(BertPooler, self).__init__() - self.dense_act = LinearActivation(config.hidden_size, - config.hidden_size, - act="tanh") - - def forward(self, hidden_states): - # We "pool" the model by simply taking the hidden state corresponding - # to the first token. - first_token_tensor = hidden_states[:, 0] - pooled_output = self.dense_act(first_token_tensor) - return pooled_output - - -class BertPredictionHeadTransform(nn.Module): - def __init__(self, config): - super(BertPredictionHeadTransform, self).__init__() - self.dense_act = LinearActivation(config.hidden_size, - config.hidden_size, - act=config.hidden_act) - self.LayerNorm = BertLayerNorm(config.hidden_size, eps=1e-12) - - def forward(self, hidden_states): - hidden_states = self.dense_act(hidden_states) - hidden_states = self.LayerNorm(hidden_states) - return hidden_states - - -class BertLMPredictionHead(nn.Module): - def __init__(self, config, bert_model_embedding_weights): - super(BertLMPredictionHead, self).__init__() - self.transform = BertPredictionHeadTransform(config) - - # The output weights are the same as the input embeddings, but there is - # an output-only bias for each token. - self.decoder = nn.Linear(bert_model_embedding_weights.size(1), - bert_model_embedding_weights.size(0), - bias=False) - self.decoder.weight = bert_model_embedding_weights - self.bias = nn.Parameter( - torch.zeros(bert_model_embedding_weights.size(0))) - - def forward(self, hidden_states, masked_token_indexes): - hidden_states = self.transform(hidden_states) - - if masked_token_indexes is not None: - hidden_states = torch.index_select( - hidden_states.view(-1, hidden_states.shape[-1]), 0, - masked_token_indexes) - - torch.cuda.nvtx.range_push( - "decoder input.size() = {}, weight.size() = {}".format( - hidden_states.size(), self.decoder.weight.size())) - hidden_states = self.decoder(hidden_states) + self.bias - torch.cuda.nvtx.range_pop() - return hidden_states - - -class BertOnlyMLMHead(nn.Module): - def __init__(self, config, bert_model_embedding_weights): - super(BertOnlyMLMHead, self).__init__() - self.predictions = BertLMPredictionHead(config, - bert_model_embedding_weights) - - def forward(self, sequence_output): - prediction_scores = self.predictions(sequence_output) - return prediction_scores - - -class BertOnlyNSPHead(nn.Module): - def __init__(self, config): - super(BertOnlyNSPHead, self).__init__() - self.seq_relationship = nn.Linear(config.hidden_size, 2) - - def forward(self, pooled_output): - seq_relationship_score = self.seq_relationship(pooled_output) - return seq_relationship_score - - -class BertPreTrainingHeads(nn.Module): - def __init__(self, config, bert_model_embedding_weights): - super(BertPreTrainingHeads, self).__init__() - self.predictions = BertLMPredictionHead(config, - bert_model_embedding_weights) - self.seq_relationship = nn.Linear(config.hidden_size, 2) - - def forward(self, - sequence_output, - pooled_output, - masked_token_indexes=None): - prediction_scores = self.predictions(sequence_output, - masked_token_indexes) - seq_relationship_score = self.seq_relationship(pooled_output) - return prediction_scores, seq_relationship_score - - -class BertPreTrainedModel(nn.Module): - """ An abstract class to handle weights initialization and - a simple interface for dowloading and loading pretrained models. - """ - def __init__(self, config, *inputs, **kwargs): - super(BertPreTrainedModel, self).__init__() - if not isinstance(config, BertConfig): - raise ValueError( - "Parameter config in `{}(config)` should be an instance of class `BertConfig`. " - "To create a model from a Google pretrained model use " - "`model = {}.from_pretrained(PRETRAINED_MODEL_NAME)`".format( - self.__class__.__name__, self.__class__.__name__)) - self.config = config - - def init_bert_weights(self, module): - """ Initialize the weights. - """ - if isinstance(module, (nn.Linear, nn.Embedding)): - # Slightly different from the TF version which uses truncated_normal for initialization - # cf https://github.com/pytorch/pytorch/pull/5617 - num_layers = self.config.num_hidden_layers - std = self.config.initializer_range - if hasattr(module, 'bert_output_layer'): - # "Accounting for accumulation on the residual path" - #print("Accounting for accumulation on the residual path") - std = self.config.initializer_range / math.sqrt( - 2.0 * num_layers) - module.weight.data.normal_(mean=0.0, std=std) - elif isinstance(module, BertLayerNorm): - module.bias.data.zero_() - module.weight.data.fill_(1.0) - if isinstance(module, nn.Linear) and module.bias is not None: - module.bias.data.zero_() - - @classmethod - def from_pretrained(cls, - pretrained_model_name_or_path, - state_dict=None, - cache_dir=None, - from_tf=False, - *inputs, - **kwargs): - """ - Instantiate a BertPreTrainedModel from a pre-trained model file or a pytorch state dict. - Download and cache the pre-trained model file if needed. - - Params: - pretrained_model_name_or_path: either: - - a str with the name of a pre-trained model to load selected in the list of: - . `bert-base-uncased` - . `bert-large-uncased` - . `bert-base-cased` - . `bert-large-cased` - . `bert-base-multilingual-uncased` - . `bert-base-multilingual-cased` - . `bert-base-chinese` - - a path or url to a pretrained model archive containing: - . `bert_config.json` a configuration file for the model - . `pytorch_model.bin` a PyTorch dump of a BertForPreTraining instance - - a path or url to a pretrained model archive containing: - . `bert_config.json` a configuration file for the model - . `model.chkpt` a TensorFlow checkpoint - from_tf: should we load the weights from a locally saved TensorFlow checkpoint - cache_dir: an optional path to a folder in which the pre-trained models will be cached. - state_dict: an optional state dictionnary (collections.OrderedDict object) to use instead of Google pre-trained models - *inputs, **kwargs: additional input for the specific Bert class - (ex: num_labels for BertForSequenceClassification) - """ - if pretrained_model_name_or_path in PRETRAINED_MODEL_ARCHIVE_MAP: - archive_file = PRETRAINED_MODEL_ARCHIVE_MAP[ - pretrained_model_name_or_path] - else: - archive_file = pretrained_model_name_or_path - # redirect to the cache, if necessary - try: - resolved_archive_file = cached_path(archive_file, - cache_dir=cache_dir) - except EnvironmentError: - logger.error( - "Model name '{}' was not found in model name list ({}). " - "We assumed '{}' was a path or url but couldn't find any file " - "associated to this path or url.".format( - pretrained_model_name_or_path, - ', '.join(PRETRAINED_MODEL_ARCHIVE_MAP.keys()), - archive_file)) - return None - if resolved_archive_file == archive_file: - logger.info("loading archive file {}".format(archive_file)) - else: - logger.info("loading archive file {} from cache at {}".format( - archive_file, resolved_archive_file)) - tempdir = None - if os.path.isdir(resolved_archive_file) or from_tf: - serialization_dir = resolved_archive_file - else: - # Extract archive to temp dir - tempdir = tempfile.mkdtemp() - logger.info("extracting archive file {} to temp dir {}".format( - resolved_archive_file, tempdir)) - with tarfile.open(resolved_archive_file, 'r:gz') as archive: - archive.extractall(tempdir) - serialization_dir = tempdir - # Load config - config_file = os.path.join(serialization_dir, CONFIG_NAME) - config = BertConfig.from_json_file(config_file) - logger.info("Model config {}".format(config)) - # Instantiate model. - model = cls(config, *inputs, **kwargs) - if state_dict is None and not from_tf: - weights_path = os.path.join(serialization_dir, WEIGHTS_NAME) - state_dict = torch.load( - weights_path, - map_location='cpu' if not torch.cuda.is_available() else None) - if tempdir: - # Clean up temp dir - shutil.rmtree(tempdir) - if from_tf: - # Directly load from a TensorFlow checkpoint - weights_path = os.path.join(serialization_dir, TF_WEIGHTS_NAME) - return load_tf_weights_in_bert(model, weights_path) - # Load from a PyTorch state_dict - old_keys = [] - new_keys = [] - for key in state_dict.keys(): - new_key = None - if 'gamma' in key: - new_key = key.replace('gamma', 'weight') - if 'beta' in key: - new_key = key.replace('beta', 'bias') - if new_key: - old_keys.append(key) - new_keys.append(new_key) - for old_key, new_key in zip(old_keys, new_keys): - state_dict[new_key] = state_dict.pop(old_key) - - missing_keys = [] - unexpected_keys = [] - error_msgs = [] - # copy state_dict so _load_from_state_dict can modify it - metadata = getattr(state_dict, '_metadata', None) - state_dict = state_dict.copy() - if metadata is not None: - state_dict._metadata = metadata - - def load(module, prefix=''): - local_metadata = {} if metadata is None else metadata.get( - prefix[:-1], {}) - module._load_from_state_dict(state_dict, prefix, local_metadata, - True, missing_keys, unexpected_keys, - error_msgs) - for name, child in module._modules.items(): - if child is not None: - load(child, prefix + name + '.') - - start_prefix = '' - if not hasattr(model, 'bert') and any( - s.startswith('bert.') for s in state_dict.keys()): - start_prefix = 'bert.' - load(model, prefix=start_prefix) - if len(missing_keys) > 0: - logger.info( - "Weights of {} not initialized from pretrained model: {}". - format(model.__class__.__name__, missing_keys)) - if len(unexpected_keys) > 0: - logger.info( - "Weights from pretrained model not used in {}: {}".format( - model.__class__.__name__, unexpected_keys)) - if len(error_msgs) > 0: - raise RuntimeError( - 'Error(s) in loading state_dict for {}:\n\t{}'.format( - model.__class__.__name__, "\n\t".join(error_msgs))) - return model - - -class BertModel(BertPreTrainedModel): - """BERT model ("Bidirectional Embedding Representations from a Transformer"). - - Params: - config: a BertConfig class instance with the configuration to build a new model - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `output_all_encoded_layers`: boolean which controls the content of the `encoded_layers` output as described below. Default: `True`. - - Outputs: Tuple of (encoded_layers, pooled_output) - `encoded_layers`: controled by `output_all_encoded_layers` argument: - - `output_all_encoded_layers=True`: outputs a list of the full sequences of encoded-hidden-states at the end - of each attention block (i.e. 12 full sequences for BERT-base, 24 for BERT-large), each - encoded-hidden-state is a torch.FloatTensor of size [batch_size, sequence_length, hidden_size], - - `output_all_encoded_layers=False`: outputs only the full sequence of hidden-states corresponding - to the last attention block of shape [batch_size, sequence_length, hidden_size], - `pooled_output`: a torch.FloatTensor of size [batch_size, hidden_size] which is the output of a - classifier pretrained on top of the hidden state associated to the first character of the - input (`CLS`) to train on the Next-Sentence task (see BERT's paper). - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = modeling.BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - model = modeling.BertModel(config=config) - all_encoder_layers, pooled_output = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config, args=None): - super(BertModel, self).__init__(config) - self.embeddings = BertEmbeddings(config) - # set pad_token_id that is used for sparse attention padding - self.pad_token_id = config.pad_token_id if hasattr( - config, 'pad_token_id') and config.pad_token_id is not None else 0 - # set sparse_attention_config if it has been selected - self.sparse_attention_config = get_sparse_attention_config( - args, config.num_attention_heads) - self.sparse_attention_utils = get_sparse_attention_utils( - self.sparse_attention_config) - self.encoder = BertEncoder( - config, args, sparse_attention_config=self.sparse_attention_config) - self.pooler = BertPooler(config) - self.apply(self.init_bert_weights) - logger.info("Init BERT pretrain model") - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - output_all_encoded_layers=True, - checkpoint_activations=False, - progressive_layer_drop=False, - theta=0.5): - if attention_mask is None: - attention_mask = torch.ones_like(input_ids) - if token_type_ids is None: - token_type_ids = torch.zeros_like(input_ids) - - # We create a 3D attention mask from a 2D tensor mask. - # Sizes are [batch_size, 1, 1, to_seq_length] - # So we can broadcast to [batch_size, num_heads, from_seq_length, to_seq_length] - # this attention mask is more simple than the triangular masking of causal attention - # used in OpenAI GPT, we just need to prepare the broadcast dimension here. - extended_attention_mask = attention_mask.unsqueeze(1).unsqueeze(2) - - # Since attention_mask is 1.0 for positions we want to attend and 0.0 for - # masked positions, this operation will create a tensor which is 0.0 for - # positions we want to attend and -10000.0 for masked positions. - # Since we are adding it to the raw scores before the softmax, this is - # effectively the same as removing these entirely. - extended_attention_mask = extended_attention_mask.to( - dtype=next(self.parameters()).dtype) # fp16 compatibility - extended_attention_mask = (1.0 - extended_attention_mask) * -10000.0 - - # If BertEncoder uses sparse attention, it needs to be padded based on the sparse attention block size - if self.sparse_attention_config is not None: - pad_len, input_ids, attention_mask, token_type_ids, position_ids, inputs_embeds = self.sparse_attention_utils.pad_to_block_size( - block_size=self.sparse_attention_config.block, - input_ids=input_ids, - attention_mask=extended_attention_mask, - token_type_ids=token_type_ids, - position_ids=None, - inputs_embeds=None, - pad_token_id=self.pad_token_id, - model_mbeddings=self.embeddings) - - embedding_output = self.embeddings(input_ids, token_type_ids) - encoded_layers = self.encoder( - embedding_output, - extended_attention_mask, - output_all_encoded_layers=output_all_encoded_layers, - checkpoint_activations=checkpoint_activations, - progressive_layer_drop=progressive_layer_drop, - theta=theta) - sequence_output = encoded_layers[-1] - pooled_output = self.pooler(sequence_output) - - # If BertEncoder uses sparse attention, and input_ids were padded, sequence output needs to be unpadded to original length - if self.sparse_attention_config is not None and pad_len > 0: - encoded_layers[ - -1] = self.sparse_attention_utils.unpad_sequence_output( - pad_len, encoded_layers[-1]) - - if not output_all_encoded_layers: - encoded_layers = encoded_layers[-1] - return encoded_layers, pooled_output - - -class BertForPreTrainingPreLN(BertPreTrainedModel): - """BERT model with pre-training heads. - This module comprises the BERT model followed by the two pre-training heads: - - the masked language modeling head, and - - the next sentence classification head. - - Params: - config: a BertConfig class instance with the configuration to build a new model. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `masked_lm_labels`: optional masked language modeling labels: torch.LongTensor of shape [batch_size, sequence_length] - with indices selected in [-1, 0, ..., vocab_size]. All labels set to -1 are ignored (masked), the loss - is only computed for the labels set in [0, ..., vocab_size] - `next_sentence_label`: optional next sentence classification loss: torch.LongTensor of shape [batch_size] - with indices selected in [0, 1]. - 0 => next sentence is the continuation, 1 => next sentence is a random sentence. - - Outputs: - if `masked_lm_labels` and `next_sentence_label` are not `None`: - Outputs the total_loss which is the sum of the masked language modeling loss and the next - sentence classification loss. - if `masked_lm_labels` or `next_sentence_label` is `None`: - Outputs a tuple comprising - - the masked language modeling logits of shape [batch_size, sequence_length, vocab_size], and - - the next sentence classification logits of shape [batch_size, 2]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - model = BertForPreTraining(config) - masked_lm_logits_scores, seq_relationship_logits = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config, args): - super(BertForPreTrainingPreLN, self).__init__(config) - self.bert = BertModel(config, args) - self.cls = BertPreTrainingHeads( - config, self.bert.embeddings.word_embeddings.weight) - self.apply(self.init_bert_weights) - self.args = args - - def forward(self, batch, **kwargs): - progressive_layer_drop = kwargs.get('progressive_layer_drop', False) - theta = kwargs.get('pld_theta', 1.0) - - input_ids = batch[1] - token_type_ids = batch[3] - attention_mask = batch[2] - masked_lm_labels = batch[5] - next_sentence_label = batch[4] - checkpoint_activations = False - - sequence_output, pooled_output = self.bert( - input_ids, - token_type_ids, - attention_mask, - output_all_encoded_layers=False, - checkpoint_activations=checkpoint_activations, - progressive_layer_drop=progressive_layer_drop, - theta=theta) - - if masked_lm_labels is not None and next_sentence_label is not None: - # filter out all masked labels. - masked_token_indexes = torch.nonzero( - (masked_lm_labels + 1).view(-1)).view(-1) - prediction_scores, seq_relationship_score = self.cls( - sequence_output, pooled_output, masked_token_indexes) - target = torch.index_select(masked_lm_labels.view(-1), 0, - masked_token_indexes) - - loss_fct = CrossEntropyLoss(ignore_index=-1) - masked_lm_loss = loss_fct( - prediction_scores.view(-1, self.config.vocab_size), target) - next_sentence_loss = loss_fct(seq_relationship_score.view(-1, 2), - next_sentence_label.view(-1)) - total_loss = masked_lm_loss + next_sentence_loss - return total_loss - else: - prediction_scores, seq_relationship_score = self.cls( - sequence_output, pooled_output) - return prediction_scores, seq_relationship_score - - -class BertForMaskedLM(BertPreTrainedModel): - """BERT model with the masked language modeling head. - This module comprises the BERT model followed by the masked language modeling head. - - Params: - config: a BertConfig class instance with the configuration to build a new model. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `masked_lm_labels`: masked language modeling labels: torch.LongTensor of shape [batch_size, sequence_length] - with indices selected in [-1, 0, ..., vocab_size]. All labels set to -1 are ignored (masked), the loss - is only computed for the labels set in [0, ..., vocab_size] - - Outputs: - if `masked_lm_labels` is not `None`: - Outputs the masked language modeling loss. - if `masked_lm_labels` is `None`: - Outputs the masked language modeling logits of shape [batch_size, sequence_length, vocab_size]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - model = BertForMaskedLM(config) - masked_lm_logits_scores = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config): - super(BertForMaskedLM, self).__init__(config) - self.bert = BertModel(config) - self.cls = BertOnlyMLMHead(config, - self.bert.embeddings.word_embeddings.weight) - self.apply(self.init_bert_weights) - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - masked_lm_labels=None, - checkpoint_activations=False): - sequence_output, _ = self.bert(input_ids, - token_type_ids, - attention_mask, - output_all_encoded_layers=False) - prediction_scores = self.cls(sequence_output) - - if masked_lm_labels is not None: - loss_fct = CrossEntropyLoss(ignore_index=-1) - masked_lm_loss = loss_fct( - prediction_scores.view(-1, self.config.vocab_size), - masked_lm_labels.view(-1)) - return masked_lm_loss - else: - return prediction_scores - - -class BertForNextSentencePrediction(BertPreTrainedModel): - """BERT model with next sentence prediction head. - This module comprises the BERT model followed by the next sentence classification head. - - Params: - config: a BertConfig class instance with the configuration to build a new model. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `next_sentence_label`: next sentence classification loss: torch.LongTensor of shape [batch_size] - with indices selected in [0, 1]. - 0 => next sentence is the continuation, 1 => next sentence is a random sentence. - - Outputs: - if `next_sentence_label` is not `None`: - Outputs the total_loss which is the sum of the masked language modeling loss and the next - sentence classification loss. - if `next_sentence_label` is `None`: - Outputs the next sentence classification logits of shape [batch_size, 2]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - model = BertForNextSentencePrediction(config) - seq_relationship_logits = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config): - super(BertForNextSentencePrediction, self).__init__(config) - self.bert = BertModel(config) - self.cls = BertOnlyNSPHead(config) - self.apply(self.init_bert_weights) - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - next_sentence_label=None, - checkpoint_activations=False): - _, pooled_output = self.bert(input_ids, - token_type_ids, - attention_mask, - output_all_encoded_layers=False) - seq_relationship_score = self.cls(pooled_output) - - if next_sentence_label is not None: - loss_fct = CrossEntropyLoss(ignore_index=-1) - next_sentence_loss = loss_fct(seq_relationship_score.view(-1, 2), - next_sentence_label.view(-1)) - return next_sentence_loss - else: - return seq_relationship_score - - -class BertForSequenceClassification(BertPreTrainedModel): - """BERT model for classification. - This module is composed of the BERT model with a linear layer on top of - the pooled output. - - Params: - `config`: a BertConfig class instance with the configuration to build a new model. - `num_labels`: the number of classes for the classifier. Default = 2. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `labels`: labels for the classification output: torch.LongTensor of shape [batch_size] - with indices selected in [0, ..., num_labels]. - - Outputs: - if `labels` is not `None`: - Outputs the CrossEntropy classification loss of the output with the labels. - if `labels` is `None`: - Outputs the classification logits of shape [batch_size, num_labels]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - num_labels = 2 - - model = BertForSequenceClassification(config, num_labels) - logits = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, args, config, num_labels): - super(BertForSequenceClassification, self).__init__(config) - self.num_labels = num_labels - self.bert = BertModel(config, args) - self.dropout = nn.Dropout(config.hidden_dropout_prob) - self.classifier = nn.Linear(config.hidden_size, num_labels) - self.apply(self.init_bert_weights) - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - labels=None, - checkpoint_activations=False): - _, pooled_output = self.bert(input_ids, - token_type_ids, - attention_mask, - output_all_encoded_layers=False) - pooled_output = self.dropout(pooled_output) - logits = self.classifier(pooled_output) - - if labels is not None: - loss_fct = CrossEntropyLoss() - loss = loss_fct(logits.view(-1, self.num_labels), labels.view(-1)) - return loss - else: - return logits - - -class BertForMultipleChoice(BertPreTrainedModel): - """BERT model for multiple choice tasks. - This module is composed of the BERT model with a linear layer on top of - the pooled output. - - Params: - `config`: a BertConfig class instance with the configuration to build a new model. - `num_choices`: the number of classes for the classifier. Default = 2. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, num_choices, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, num_choices, sequence_length] - with the token types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` - and type 1 corresponds to a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, num_choices, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `labels`: labels for the classification output: torch.LongTensor of shape [batch_size] - with indices selected in [0, ..., num_choices]. - - Outputs: - if `labels` is not `None`: - Outputs the CrossEntropy classification loss of the output with the labels. - if `labels` is `None`: - Outputs the classification logits of shape [batch_size, num_labels]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[[31, 51, 99], [15, 5, 0]], [[12, 16, 42], [14, 28, 57]]]) - input_mask = torch.LongTensor([[[1, 1, 1], [1, 1, 0]],[[1,1,0], [1, 0, 0]]]) - token_type_ids = torch.LongTensor([[[0, 0, 1], [0, 1, 0]],[[0, 1, 1], [0, 0, 1]]]) - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - num_choices = 2 - - model = BertForMultipleChoice(config, num_choices) - logits = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config, num_choices): - super(BertForMultipleChoice, self).__init__(config) - self.num_choices = num_choices - self.bert = BertModel(config) - self.dropout = nn.Dropout(config.hidden_dropout_prob) - self.classifier = nn.Linear(config.hidden_size, 1) - self.apply(self.init_bert_weights) - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - labels=None, - checkpoint_activations=False): - flat_input_ids = input_ids.view(-1, input_ids.size(-1)) - flat_token_type_ids = token_type_ids.view(-1, token_type_ids.size(-1)) - flat_attention_mask = attention_mask.view(-1, attention_mask.size(-1)) - _, pooled_output = self.bert(flat_input_ids, - flat_token_type_ids, - flat_attention_mask, - output_all_encoded_layers=False) - pooled_output = self.dropout(pooled_output) - logits = self.classifier(pooled_output) - reshaped_logits = logits.view(-1, self.num_choices) - - if labels is not None: - loss_fct = CrossEntropyLoss() - loss = loss_fct(reshaped_logits, labels) - return loss - else: - return reshaped_logits - - -class BertForTokenClassification(BertPreTrainedModel): - """BERT model for token-level classification. - This module is composed of the BERT model with a linear layer on top of - the full hidden state of the last layer. - - Params: - `config`: a BertConfig class instance with the configuration to build a new model. - `num_labels`: the number of classes for the classifier. Default = 2. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `labels`: labels for the classification output: torch.LongTensor of shape [batch_size, sequence_length] - with indices selected in [0, ..., num_labels]. - - Outputs: - if `labels` is not `None`: - Outputs the CrossEntropy classification loss of the output with the labels. - if `labels` is `None`: - Outputs the classification logits of shape [batch_size, sequence_length, num_labels]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - num_labels = 2 - - model = BertForTokenClassification(config, num_labels) - logits = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config, num_labels): - super(BertForTokenClassification, self).__init__(config) - self.num_labels = num_labels - self.bert = BertModel(config) - self.dropout = nn.Dropout(config.hidden_dropout_prob) - self.classifier = nn.Linear(config.hidden_size, num_labels) - self.apply(self.init_bert_weights) - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - labels=None, - checkpoint_activations=False): - sequence_output, _ = self.bert(input_ids, - token_type_ids, - attention_mask, - output_all_encoded_layers=False) - sequence_output = self.dropout(sequence_output) - logits = self.classifier(sequence_output) - - if labels is not None: - loss_fct = CrossEntropyLoss() - # Only keep active parts of the loss - if attention_mask is not None: - active_loss = attention_mask.view(-1) == 1 - active_logits = logits.view(-1, self.num_labels)[active_loss] - active_labels = labels.view(-1)[active_loss] - loss = loss_fct(active_logits, active_labels) - else: - loss = loss_fct(logits.view(-1, self.num_labels), - labels.view(-1)) - return loss - else: - return logits - - -class BertForQuestionAnswering(BertPreTrainedModel): - """BERT model for Question Answering (span extraction). - This module is composed of the BERT model with a linear layer on top of - the sequence output that computes start_logits and end_logits - - Params: - `config`: a BertConfig class instance with the configuration to build a new model. - - Inputs: - `input_ids`: a torch.LongTensor of shape [batch_size, sequence_length] - with the word token indices in the vocabulary(see the tokens preprocessing logic in the scripts - `extract_features.py`, `run_classifier.py` and `run_squad.py`) - `token_type_ids`: an optional torch.LongTensor of shape [batch_size, sequence_length] with the token - types indices selected in [0, 1]. Type 0 corresponds to a `sentence A` and type 1 corresponds to - a `sentence B` token (see BERT paper for more details). - `attention_mask`: an optional torch.LongTensor of shape [batch_size, sequence_length] with indices - selected in [0, 1]. It's a mask to be used if the input sequence length is smaller than the max - input sequence length in the current batch. It's the mask that we typically use for attention when - a batch has varying length sentences. - `start_positions`: position of the first token for the labeled span: torch.LongTensor of shape [batch_size]. - Positions are clamped to the length of the sequence and position outside of the sequence are not taken - into account for computing the loss. - `end_positions`: position of the last token for the labeled span: torch.LongTensor of shape [batch_size]. - Positions are clamped to the length of the sequence and position outside of the sequence are not taken - into account for computing the loss. - - Outputs: - if `start_positions` and `end_positions` are not `None`: - Outputs the total_loss which is the sum of the CrossEntropy loss for the start and end token positions. - if `start_positions` or `end_positions` is `None`: - Outputs a tuple of start_logits, end_logits which are the logits respectively for the start and end - position tokens of shape [batch_size, sequence_length]. - - Example usage: - ```python - # Already been converted into WordPiece token ids - input_ids = torch.LongTensor([[31, 51, 99], [15, 5, 0]]) - input_mask = torch.LongTensor([[1, 1, 1], [1, 1, 0]]) - token_type_ids = torch.LongTensor([[0, 0, 1], [0, 1, 0]]) - - config = BertConfig(vocab_size_or_config_json_file=32000, hidden_size=768, - num_hidden_layers=12, num_attention_heads=12, intermediate_size=3072) - - model = BertForQuestionAnswering(config) - start_logits, end_logits = model(input_ids, token_type_ids, input_mask) - ``` - """ - def __init__(self, config): - super(BertForQuestionAnswering, self).__init__(config) - self.bert = BertModel(config) - # TODO check with Google if it's normal there is no dropout on the token classifier of SQuAD in the TF version - # self.dropout = nn.Dropout(config.hidden_dropout_prob) - self.qa_outputs = nn.Linear(config.hidden_size, 2) - self.apply(self.init_bert_weights) - - def forward(self, - input_ids, - token_type_ids=None, - attention_mask=None, - start_positions=None, - end_positions=None, - checkpoint_activations=False): - sequence_output, _ = self.bert(input_ids, - token_type_ids, - attention_mask, - output_all_encoded_layers=False) - logits = self.qa_outputs(sequence_output) - start_logits, end_logits = logits.split(1, dim=-1) - start_logits = start_logits.squeeze(-1) - end_logits = end_logits.squeeze(-1) - - if start_positions is not None and end_positions is not None: - # If we are on multi-GPU, split add a dimension - if len(start_positions.size()) > 1: - start_positions = start_positions.squeeze(-1) - if len(end_positions.size()) > 1: - end_positions = end_positions.squeeze(-1) - # sometimes the start/end positions are outside our model inputs, we ignore these terms - ignored_index = start_logits.size(1) - start_positions.clamp_(0, ignored_index) - end_positions.clamp_(0, ignored_index) - - loss_fct = CrossEntropyLoss(ignore_index=ignored_index) - start_loss = loss_fct(start_logits, start_positions) - end_loss = loss_fct(end_logits, end_positions) - total_loss = (start_loss + end_loss) / 2 - return total_loss - else: - return start_logits, end_logits diff --git a/training/bing_bert/run_glue_bert_base_finetune.sh b/training/bing_bert/run_glue_bert_base_finetune.sh index 7a1c12cfe..628c03f1e 100755 --- a/training/bing_bert/run_glue_bert_base_finetune.sh +++ b/training/bing_bert/run_glue_bert_base_finetune.sh @@ -48,7 +48,7 @@ run_cmd="python3.6 -m torch.distributed.launch \ --learning_rate ${LR} \ --num_train_epochs ${NUM_EPOCH} \ --output_dir ${OUTPUT_DIR}_${TASK} \ - --progressive_layer_drop \ + --preln \ --model_file $CHECKPOINT_PATH &> $LOG_DIR/${model_name}/${JOBNAME}_${TASK}_bzs${EFFECTIVE_BATCH_SIZE}_lr${LR}_epoch${NUM_EPOCH}.txt " echo ${run_cmd} diff --git a/training/bing_bert/run_glue_classifier_bert_base.py b/training/bing_bert/run_glue_classifier_bert_base.py index 9f46c8a6d..1bd2b8d27 100755 --- a/training/bing_bert/run_glue_classifier_bert_base.py +++ b/training/bing_bert/run_glue_classifier_bert_base.py @@ -679,18 +679,10 @@ def main(): help="Whether to use Focal Loss for finetuning.") parser.add_argument('--gamma', type=float, default=0.5, help="Gamma parameter to be used in focal loss.") - parser.add_argument('--deepspeed_sparse_attention', - default=False, - action='store_true', - help='Use DeepSpeed sparse self attention.') parser.add_argument('--deepspeed_transformer_kernel', default=False, action='store_true', help='Use DeepSpeed transformer kernel to accelerate.') - parser.add_argument('--progressive_layer_drop', - default=False, - action='store_true', - help="Whether to enable progressive layer dropping or not") parser.add_argument( '--preln', action='store_true', @@ -815,10 +807,7 @@ def main(): "initializer_range": 0.02 } - if args.progressive_layer_drop: - print("BertBaseConfigPreLnLayerDrop") - from nvidia.modelingpreln_layerdrop import BertForSequenceClassification, BertConfig - elif args.preln: + if args.preln: from nvidia.modelingpreln import BertForSequenceClassification, BertConfig, BertLayer else: from nvidia.modeling import BertForSequenceClassification, BertConfig, BertLayer diff --git a/training/bing_bert/run_glue_classifier_bert_large.py b/training/bing_bert/run_glue_classifier_bert_large.py index 7d2352d61..33bf8a861 100755 --- a/training/bing_bert/run_glue_classifier_bert_large.py +++ b/training/bing_bert/run_glue_classifier_bert_large.py @@ -751,10 +751,6 @@ def main(): type=float, default=0.5, help="Gamma parameter to be used in focal loss.") - parser.add_argument('--deepspeed_sparse_attention', - default=False, - action='store_true', - help='Use DeepSpeed sparse self attention.') parser.add_argument( '--preln', action='store_true', @@ -766,11 +762,6 @@ def main(): default=False, action='store_true', help='Use DeepSpeed transformer kernel to accelerate.') - parser.add_argument( - '--progressive_layer_drop', - default=False, - action='store_true', - help="Whether to enable progressive layer dropping or not") parser = deepspeed.add_config_arguments(parser) args = parser.parse_args() @@ -893,10 +884,7 @@ def main(): "initializer_range": 0.02 } - if args.progressive_layer_drop: - print("BertBaseConfigPreLnLayerDrop") - from nvidia.modelingpreln_layerdrop import BertForSequenceClassification, BertConfig, BertLayer - elif args.preln: + if args.preln: from nvidia.modelingpreln import BertForSequenceClassification, BertConfig, BertLayer else: from nvidia.modeling import BertForSequenceClassification, BertConfig, BertLayer diff --git a/training/bing_bert/turing/models.py b/training/bing_bert/turing/models.py index 35a8d202f..e9ec3e3f6 100755 --- a/training/bing_bert/turing/models.py +++ b/training/bing_bert/turing/models.py @@ -105,12 +105,7 @@ def __init__(self, args): self.config = args.config if not args.use_pretrain: - - if args.progressive_layer_drop: - print("BertConfigPreLnLayerDrop") - from nvidia.modelingpreln_layerdrop import BertForPreTrainingPreLN, BertConfig - else: - from nvidia.modelingpreln import BertForPreTrainingPreLN, BertConfig + from nvidia.modelingpreln import BertForPreTrainingPreLN, BertConfig bert_config = BertConfig(**self.config["bert_model_config"]) bert_config.vocab_size = len(args.tokenizer.vocab) diff --git a/training/bing_bert/utils.py b/training/bing_bert/utils.py index fd6869463..83012ed83 100755 --- a/training/bing_bert/utils.py +++ b/training/bing_bert/utils.py @@ -186,21 +186,11 @@ def get_argument_parser(): help= 'Use DeepSpeed transformer kernel memory optimization to checkpoint GELU activation.' ) - parser.add_argument('--deepspeed_sparse_attention', - default=False, - action='store_true', - help='Use DeepSpeed sparse self attention.') - parser.add_argument('--use_nvidia_dataset', default=False, action='store_true', help='Use Nvidia pretraining dataset.') - parser.add_argument('--progressive_layer_drop', - default=False, - action='store_true', - help="Whether to enable progressive layer dropping or not") - return parser From 694c45797fa395d27629e920cb4cabb488e1e927 Mon Sep 17 00:00:00 2001 From: Hongwei Chen Date: Sat, 26 Sep 2026 17:49:30 +0000 Subject: [PATCH 2/3] Fix error Signed-off-by: Hongwei Chen --- training/bing_bert/deepspeed_train.py | 4 +- training/bing_bert/nvidia/modelingpreln.py | 9 +- .../bing_bert/run_glue_bert_base_finetune.sh | 9 +- .../bing_bert/run_glue_bert_large_finetune.sh | 9 +- .../run_glue_classifier_bert_base.py | 100 +++++++----------- .../run_glue_classifier_bert_large.py | 97 +++++++---------- 6 files changed, 90 insertions(+), 138 deletions(-) diff --git a/training/bing_bert/deepspeed_train.py b/training/bing_bert/deepspeed_train.py index 6626a307a..da3b1371b 100755 --- a/training/bing_bert/deepspeed_train.py +++ b/training/bing_bert/deepspeed_train.py @@ -425,9 +425,7 @@ def prepare_model_optimizer(args): model.set_device(args.device) args.fp16 = model.network.fp16_enabled() args.use_lamb = (model.network.optimizer_name() == - deepspeed.runtime.config.LAMB_OPTIMIZER - or model.network.optimizer_name() == - deepspeed.runtime.config.ONEBIT_LAMB_OPTIMIZER) + deepspeed.runtime.config.LAMB_OPTIMIZER) # Prepare Summary Writer and saved_models path if dist.get_rank() == 0: diff --git a/training/bing_bert/nvidia/modelingpreln.py b/training/bing_bert/nvidia/modelingpreln.py index 87dc7f69a..9180f1a7f 100755 --- a/training/bing_bert/nvidia/modelingpreln.py +++ b/training/bing_bert/nvidia/modelingpreln.py @@ -541,7 +541,7 @@ def __init__(self, config, args): local_rank=args.local_rank if hasattr(args, 'local_rank') else -1, seed=args.seed, - fp16=ds_config.fp16_enabled, + fp16=ds_config.float16_config.enabled, pre_layer_norm=True, attn_dropout_checkpoint=args.attention_dropout_checkpoint, normalize_invertible=args.normalize_invertible, @@ -1230,6 +1230,7 @@ class BertForSequenceClassification(BertPreTrainedModel): the pooled output. Params: + `args`: the parsed command-line arguments (e.g. `args.deepspeed_transformer_kernel`). `config`: a BertConfig class instance with the configuration to build a new model. `num_labels`: the number of classes for the classifier. Default = 2. @@ -1265,14 +1266,14 @@ class BertForSequenceClassification(BertPreTrainedModel): num_labels = 2 - model = BertForSequenceClassification(config, num_labels) + model = BertForSequenceClassification(args, config, num_labels) logits = model(input_ids, token_type_ids, input_mask) ``` """ - def __init__(self, config, num_labels): + def __init__(self, args, config, num_labels): super(BertForSequenceClassification, self).__init__(config) self.num_labels = num_labels - self.bert = BertModel(config) + self.bert = BertModel(config, args=args) self.dropout = nn.Dropout(config.hidden_dropout_prob) self.classifier = nn.Linear(config.hidden_size, num_labels) self.apply(self.init_bert_weights) diff --git a/training/bing_bert/run_glue_bert_base_finetune.sh b/training/bing_bert/run_glue_bert_base_finetune.sh index 628c03f1e..e4c005fae 100755 --- a/training/bing_bert/run_glue_bert_base_finetune.sh +++ b/training/bing_bert/run_glue_bert_base_finetune.sh @@ -29,16 +29,15 @@ else GRAD_ACCUM_STEPS=$((PER_GPU_BATCH_SIZE/MAX_GPU_BATCH_SIZE)) fi +mkdir -p ${LOG_DIR}/${model_name} echo "Fine Tuning $CHECKPOINT_PATH" -run_cmd="python3.6 -m torch.distributed.launch \ - --nproc_per_node=${NGPU} \ - --master_port=${MASTER_PORT} \ - run_glue_classifier_bert_base.py \ +run_cmd="deepspeed --num_gpus=${NGPU} \ + ${SCRIPT_DIR}/run_glue_classifier_bert_base.py \ --task_name $TASK \ --do_train \ --do_eval \ --deepspeed \ - --deepspeed_config ${base_dir}/glue_bert_base.json \ + --deepspeed_config ${SCRIPT_DIR}/glue_bert_base.json \ --do_lower_case \ --data_dir $GLUE_DIR/$TASK/ \ --bert_model bert-large-uncased \ diff --git a/training/bing_bert/run_glue_bert_large_finetune.sh b/training/bing_bert/run_glue_bert_large_finetune.sh index 4a63adb57..fd2407ef7 100755 --- a/training/bing_bert/run_glue_bert_large_finetune.sh +++ b/training/bing_bert/run_glue_bert_large_finetune.sh @@ -29,11 +29,10 @@ else GRAD_ACCUM_STEPS=$((PER_GPU_BATCH_SIZE/MAX_GPU_BATCH_SIZE)) fi +mkdir -p ${LOG_DIR}/${model_name} echo "Fine Tuning $CHECKPOINT_PATH" -run_cmd="python3.6 -m torch.distributed.launch \ - --nproc_per_node=${NGPU} \ - --master_port=12346 \ - run_glue_classifier_bert_large.py \ +run_cmd="deepspeed --num_gpus=${NGPU} --master_port=12346 \ + ${SCRIPT_DIR}/run_glue_classifier_bert_large.py \ --task_name $TASK \ --do_train \ --do_eval \ @@ -41,7 +40,7 @@ run_cmd="python3.6 -m torch.distributed.launch \ --deepspeed_transformer_kernel \ --fp16 \ --preln \ - --deepspeed_config ${base_dir}/glue_bert_large.json \ + --deepspeed_config ${SCRIPT_DIR}/glue_bert_large.json \ --do_lower_case \ --data_dir $GLUE_DIR/$TASK/ \ --bert_model bert-large-uncased \ diff --git a/training/bing_bert/run_glue_classifier_bert_base.py b/training/bing_bert/run_glue_classifier_bert_base.py index 1bd2b8d27..406261e97 100755 --- a/training/bing_bert/run_glue_classifier_bert_base.py +++ b/training/bing_bert/run_glue_classifier_bert_base.py @@ -652,10 +652,6 @@ def main(): parser.add_argument('--fp16', action='store_true', help="Whether to use 16-bit float precision instead of 32-bit") - parser.add_argument('--deepscale', - default=False, - action='store_true', - help="Whether to use 16-bit float precision instead of 32-bit") parser.add_argument('--loss_scale', type=float, default=0, help="Loss scaling to improve fp16 numeric stability. Only used when fp16 set to True.\n" @@ -683,6 +679,22 @@ def main(): default=False, action='store_true', help='Use DeepSpeed transformer kernel to accelerate.') + parser.add_argument('--stochastic_mode', + default=False, + action='store_true', + help='Use stochastic mode for high-performance transformer kernel.') + parser.add_argument('--attention_dropout_checkpoint', + default=False, + action='store_true', + help='Use DeepSpeed transformer kernel memory optimization to checkpoint dropout output.') + parser.add_argument('--normalize_invertible', + default=False, + action='store_true', + help='Use DeepSpeed transformer kernel memory optimization to perform invertible normalize backpropagation.') + parser.add_argument('--gelu_checkpoint', + default=False, + action='store_true', + help='Use DeepSpeed transformer kernel memory optimization to checkpoint GELU activation.') parser.add_argument( '--preln', action='store_true', @@ -727,16 +739,12 @@ def main(): "wnli": "classification", } - if args.local_rank == -1 or args.no_cuda: - device = torch.device( - "cuda" if torch.cuda.is_available() and not args.no_cuda else "cpu") - n_gpu = torch.cuda.device_count() - else: - torch.cuda.set_device(args.local_rank) - device = torch.device("cuda", args.local_rank) - n_gpu = 1 - # Initializes the distributed backend which will take care of sychronizing nodes/GPUs - torch.distributed.init_process_group(backend='nccl') + # DeepSpeed sets up torch.distributed; launch with the `deepspeed` launcher (or torchrun). + deepspeed.init_distributed(dist_backend='nccl') + args.local_rank = int(os.environ['LOCAL_RANK']) + torch.cuda.set_device(args.local_rank) + device = torch.device("cuda", args.local_rank) + n_gpu = 1 logger.info("device: {} n_gpu: {}, distributed training: {}, 16-bits training: {}".format( device, n_gpu, bool(args.local_rank != -1), args.fp16)) @@ -808,9 +816,9 @@ def main(): } if args.preln: - from nvidia.modelingpreln import BertForSequenceClassification, BertConfig, BertLayer + from nvidia.modelingpreln import BertForSequenceClassification, BertConfig else: - from nvidia.modeling import BertForSequenceClassification, BertConfig, BertLayer + from nvidia.modeling import BertForSequenceClassification, BertConfig bert_config = BertConfig(**bert_base_model_config) bert_config.vocab_size = len(tokenizer.vocab) @@ -841,36 +849,9 @@ def main(): logger.info("USING RANDOM INITIALISATION FOR FINETUNING") model.apply(model.init_bert_weights) - if args.fp16: - model.half() + # DeepSpeed casts the model to fp16 and handles data parallelism. The old replace_transformer_layer() + # training-kernel injection no longer exists; use --deepspeed_transformer_kernel for DeepSpeed kernels. model.to(device) - if args.local_rank != -1: - try: - if args.deepscale: - print("Enabling DeepScale") - from deepscale.distributed_apex import DistributedDataParallel as DDP - else: - from apex.parallel import DistributedDataParallel as DDP - except ImportError: - raise ImportError( - "Please install apex from https://www.github.com/nvidia/apex to use distributed and fp16 training.") - - model = DDP(model) - elif n_gpu > 1: - model = torch.nn.DataParallel(model) - - # Patch model with deepspeed transformer kernel - if not args.deepspeed_transformer_kernel: - from deepspeed import replace_transformer_layer - model = deepspeed.module_inject.replace_transformer_layer( - orig_layer_impl=BertLayer, - model=model, - micro_batch_size=args.train_batch_size, - bert_config=bert_config, - seed=args.seed, - preln=True, - fp16=args.fp16, - huggingface=False) # Prepare optimizer param_optimizer = list(model.named_parameters()) @@ -891,6 +872,11 @@ def main(): model_parameters=optimizer_grouped_parameters, dist_init_required=True) + # The DeepSpeed config is the source of truth for precision, micro-batch size and gradient accumulation. + args.fp16 = model.fp16_enabled() + args.train_batch_size = model.train_micro_batch_size_per_gpu() + args.gradient_accumulation_steps = model.gradient_accumulation_steps() + global_step = 0 nb_tr_steps = 0 tr_loss = 0 @@ -927,6 +913,8 @@ def main(): train_sampler = DistributedSampler(train_data) train_dataloader = DataLoader( train_data, sampler=train_sampler, batch_size=args.train_batch_size) + num_train_optimization_steps = len(train_dataloader) // args.gradient_accumulation_steps * int( + args.num_train_epochs) model.train() for _ in trange(int(args.num_train_epochs), desc="Epoch"): @@ -951,25 +939,13 @@ def main(): loss_fct = MSELoss() loss = loss_fct(logits.view(-1), label_ids.view(-1)) - if n_gpu > 1: - loss = loss.mean() # mean() to average on multi-gpu. - if args.gradient_accumulation_steps > 1: - loss = loss / args.gradient_accumulation_steps - - if args.deepscale and args.local_rank != -1: - model.disable_need_reduction() - if (step + 1) % args.gradient_accumulation_steps == 0: - model.enable_need_reduction() - - if args.fp16: - optimizer.backward(loss) - else: - loss.backward() + # DeepSpeed scales the loss for gradient accumulation and all-reduces the gradients. + model.backward(loss) tr_loss += loss.item() nb_tr_examples += input_ids.size(0) nb_tr_steps += 1 - if (step + 1) % args.gradient_accumulation_steps == 0: + if model.is_gradient_accumulation_boundary(): if args.fp16: # modify learning rate with special warm up BERT uses # if args.fp16 is False, BertAdam is used that handles this automatically @@ -978,9 +954,9 @@ def main(): global_step/num_train_optimization_steps, args.warmup_proportion) for param_group in optimizer.param_groups: param_group['lr'] = lr_this_step - optimizer.step() - optimizer.zero_grad() global_step += 1 + # Call on every micro-step; DeepSpeed only updates weights at accumulation boundaries. + model.step() if args.do_eval and (args.local_rank == -1 or torch.distributed.get_rank() == 0): eval_examples = processor.get_dev_examples(args.data_dir) diff --git a/training/bing_bert/run_glue_classifier_bert_large.py b/training/bing_bert/run_glue_classifier_bert_large.py index 33bf8a861..5e0afc9f2 100755 --- a/training/bing_bert/run_glue_classifier_bert_large.py +++ b/training/bing_bert/run_glue_classifier_bert_large.py @@ -762,6 +762,22 @@ def main(): default=False, action='store_true', help='Use DeepSpeed transformer kernel to accelerate.') + parser.add_argument('--stochastic_mode', + default=False, + action='store_true', + help='Use stochastic mode for high-performance transformer kernel.') + parser.add_argument('--attention_dropout_checkpoint', + default=False, + action='store_true', + help='Use DeepSpeed transformer kernel memory optimization to checkpoint dropout output.') + parser.add_argument('--normalize_invertible', + default=False, + action='store_true', + help='Use DeepSpeed transformer kernel memory optimization to perform invertible normalize backpropagation.') + parser.add_argument('--gelu_checkpoint', + default=False, + action='store_true', + help='Use DeepSpeed transformer kernel memory optimization to checkpoint GELU activation.') parser = deepspeed.add_config_arguments(parser) args = parser.parse_args() @@ -799,16 +815,12 @@ def main(): "wnli": "classification", } - if args.local_rank == -1 or args.no_cuda: - device = torch.device("cuda" if torch.cuda.is_available() - and not args.no_cuda else "cpu") - n_gpu = torch.cuda.device_count() - else: - torch.cuda.set_device(args.local_rank) - device = torch.device("cuda", args.local_rank) - n_gpu = 1 - # Initializes the distributed backend which will take care of sychronizing nodes/GPUs - torch.distributed.init_process_group(backend='nccl') + # DeepSpeed sets up torch.distributed; launch with the `deepspeed` launcher (or torchrun). + deepspeed.init_distributed(dist_backend='nccl') + args.local_rank = int(os.environ['LOCAL_RANK']) + torch.cuda.set_device(args.local_rank) + device = torch.device("cuda", args.local_rank) + n_gpu = 1 logger.info( "device: {} n_gpu: {}, distributed training: {}, 16-bits training: {}". format(device, n_gpu, bool(args.local_rank != -1), args.fp16)) @@ -885,9 +897,9 @@ def main(): } if args.preln: - from nvidia.modelingpreln import BertForSequenceClassification, BertConfig, BertLayer + from nvidia.modelingpreln import BertForSequenceClassification, BertConfig else: - from nvidia.modeling import BertForSequenceClassification, BertConfig, BertLayer + from nvidia.modeling import BertForSequenceClassification, BertConfig bert_config = BertConfig(**bert_base_model_config) bert_config.vocab_size = len(tokenizer.vocab) @@ -921,37 +933,9 @@ def main(): logger.info("USING RANDOM INITIALISATION FOR FINETUNING") model.apply(model.init_bert_weights) - if args.fp16: - model.half() + # DeepSpeed casts the model to fp16 and handles data parallelism. The old replace_transformer_layer() + # training-kernel injection no longer exists; use --deepspeed_transformer_kernel for DeepSpeed kernels. model.to(device) - if args.local_rank != -1: - try: - if args.deepscale: - print("Enabling DeepScale") - from deepscale.distributed_apex import DistributedDataParallel as DDP - else: - from apex.parallel import DistributedDataParallel as DDP - except ImportError: - raise ImportError( - "Please install apex from https://www.github.com/nvidia/apex to use distributed and fp16 training." - ) - - model = DDP(model) - elif n_gpu > 1: - model = torch.nn.DataParallel(model) - - # Patch model with deepspeed transformer kernel - if not args.deepspeed_transformer_kernel: - from deepspeed import replace_transformer_layer - model = deepspeed.module_inject.replace_transformer_layer( - orig_layer_impl=BertLayer, - model=model, - micro_batch_size=args.train_batch_size, - bert_config=bert_config, - seed=args.seed, - preln=arg.preln, - fp16=args.fp16, - huggingface=False) # Prepare optimizer param_optimizer = list(model.named_parameters()) @@ -974,6 +958,11 @@ def main(): model_parameters=optimizer_grouped_parameters, dist_init_required=True) + # The DeepSpeed config is the source of truth for precision, micro-batch size and gradient accumulation. + args.fp16 = model.fp16_enabled() + args.train_batch_size = model.train_micro_batch_size_per_gpu() + args.gradient_accumulation_steps = model.gradient_accumulation_steps() + global_step = 0 nb_tr_steps = 0 tr_loss = 0 @@ -1013,6 +1002,8 @@ def main(): train_dataloader = DataLoader(train_data, sampler=train_sampler, batch_size=args.train_batch_size) + num_train_optimization_steps = len(train_dataloader) // args.gradient_accumulation_steps * int( + args.num_train_epochs) model.train() nb_tr_examples = 0 @@ -1039,25 +1030,13 @@ def main(): loss_fct = MSELoss() loss = loss_fct(logits.view(-1), label_ids.view(-1)) - if n_gpu > 1: - loss = loss.mean() # mean() to average on multi-gpu. - if args.gradient_accumulation_steps > 1: - loss = loss / args.gradient_accumulation_steps - - if args.deepscale and args.local_rank != -1: - model.disable_need_reduction() - if (step + 1) % args.gradient_accumulation_steps == 0: - model.enable_need_reduction() - - if args.fp16: - optimizer.backward(loss) - else: - loss.backward() + # DeepSpeed scales the loss for gradient accumulation and all-reduces the gradients. + model.backward(loss) tr_loss += loss.item() nb_tr_examples += input_ids.size(0) nb_tr_steps += 1 - if (step + 1) % args.gradient_accumulation_steps == 0: + if model.is_gradient_accumulation_boundary(): if args.fp16: # modify learning rate with special warm up BERT uses # if args.fp16 is False, BertAdam is used that handles this automatically @@ -1066,9 +1045,9 @@ def main(): global_step/num_train_optimization_steps, args.warmup_proportion) for param_group in optimizer.param_groups: param_group['lr'] = lr_this_step - optimizer.step() - optimizer.zero_grad() global_step += 1 + # Call on every micro-step; DeepSpeed only updates weights at accumulation boundaries. + model.step() saved_path = os.path.join(args.output_dir, "finetuned_quantized_checkpoints") From 167f9a04998d0b0513dc0e22fab00d8e15c68af8 Mon Sep 17 00:00:00 2001 From: Hongwei Chen Date: Sun, 27 Sep 2026 04:10:06 +0000 Subject: [PATCH 3/3] Fix config error Signed-off-by: Hongwei Chen --- training/BingBertGlue/nvidia/modeling.py | 2 +- training/BingBertGlue/nvidia/modelingpreln.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/training/BingBertGlue/nvidia/modeling.py b/training/BingBertGlue/nvidia/modeling.py index d887246cd..8de4f4d59 100755 --- a/training/BingBertGlue/nvidia/modeling.py +++ b/training/BingBertGlue/nvidia/modeling.py @@ -517,7 +517,7 @@ def __init__(self, config, args): initializer_range=config.initializer_range, local_rank=args.local_rank, seed=args.seed, - fp16=ds_config.fp16_enabled, + fp16=ds_config.float16_config.enabled, pre_layer_norm=False) self.layer = nn.ModuleList([ diff --git a/training/BingBertGlue/nvidia/modelingpreln.py b/training/BingBertGlue/nvidia/modelingpreln.py index db5ea1d1a..885660947 100755 --- a/training/BingBertGlue/nvidia/modelingpreln.py +++ b/training/BingBertGlue/nvidia/modelingpreln.py @@ -541,7 +541,7 @@ def __init__(self, config, args): local_rank=args.local_rank if hasattr(args, 'local_rank') else -1, seed=args.seed, - fp16=ds_config.fp16_enabled, + fp16=ds_config.float16_config.enabled, pre_layer_norm=True, # attn_dropout_checkpoint=args.attention_dropout_checkpoint, # normalize_invertible=args.normalize_invertible,