diff --git a/orttraining/orttraining/test/python/orttraining_run_glue.py b/orttraining/orttraining/test/python/orttraining_run_glue.py index 8afc3d15ae..ef20b5ad62 100644 --- a/orttraining/orttraining/test/python/orttraining_run_glue.py +++ b/orttraining/orttraining/test/python/orttraining_run_glue.py @@ -68,7 +68,6 @@ class ORTGlueTest(unittest.TestCase): self.logging_steps = 10 self.rtol = 1e-02 - def test_roberta_with_mrpc(self): expected_acc = 0.8676470588235294 expected_f1 = 0.9035714285714286 diff --git a/orttraining/orttraining/test/python/orttraining_run_multiple_choice.py b/orttraining/orttraining/test/python/orttraining_run_multiple_choice.py new file mode 100644 index 0000000000..92bb78bc3d --- /dev/null +++ b/orttraining/orttraining/test/python/orttraining_run_multiple_choice.py @@ -0,0 +1,244 @@ +# adapted from run_multiple_choice.py of huggingface transformers +# https://github.com/huggingface/transformers/blob/master/examples/multiple-choice/run_multiple_choice.py + +import dataclasses +import logging +import os +from dataclasses import dataclass, field +from typing import Dict, Optional +import unittest +import numpy as np +from numpy.testing import assert_allclose + +from transformers import ( + AutoConfig, + AutoModelForMultipleChoice, + AutoTokenizer, + EvalPrediction, + HfArgumentParser, + Trainer, + TrainingArguments, + set_seed, +) + +import onnxruntime +from onnxruntime.capi.ort_trainer import ORTTrainer, LossScaler, ModelDescription, IODescription + +from orttraining_transformer_trainer import ORTTransformerTrainer + +import torch + +from utils_multiple_choice import MultipleChoiceDataset, Split, SwagProcessor + +logger = logging.getLogger(__name__) + +def simple_accuracy(preds, labels): + return (preds == labels).mean() + +@dataclass +class ModelArguments: + """ + Arguments pertaining to which model/config/tokenizer we are going to fine-tune from. + """ + + model_name_or_path: str = field( + metadata={"help": "model identifier from huggingface.co/models"} + ) + config_name: Optional[str] = field( + default=None, metadata={"help": "Pretrained config name or path if not the same as model_name"} + ) + tokenizer_name: Optional[str] = field( + default=None, metadata={"help": "Pretrained tokenizer name or path if not the same as model_name"} + ) + cache_dir: Optional[str] = field( + default=None, metadata={"help": "Where do you want to store the pretrained models downloaded from s3"} + ) + +@dataclass +class DataTrainingArguments: + """ + Arguments pertaining to what data we are going to input our model for training and eval. + """ + + task_name: str = field(metadata={"help": "The name of the task to train on."}) + data_dir: str = field(metadata={"help": "Should contain the data files for the task."}) + max_seq_length: int = field( + default=128, + metadata={ + "help": "The maximum total input sequence length after tokenization. Sequences longer " + "than this will be truncated, sequences shorter will be padded." + }, + ) + overwrite_cache: bool = field( + default=False, metadata={"help": "Overwrite the cached training and evaluation sets"} + ) + +class ORTMultipleChoiceTest(unittest.TestCase): + + def setUp(self): + # configurations not to be changed accoss tests + self.max_seq_length = 80 + self.train_batch_size = 2 + self.eval_batch_size = 2 + self.learning_rate = 2e-5 + self.num_train_epochs = 3.0 + self.local_rank = -1 + self.overwrite_output_dir = True + self.gradient_accumulation_steps = 8 + self.data_dir = "/bert_data/hf_data/swag/swagaf/data" + self.output_dir = os.path.join(os.path.dirname(os.path.realpath(__file__)), "multiple_choice_test_output/") + self.cache_dir = '/tmp/multiple_choice/' + self.logging_steps = 10 + + def test_bert_with_swag(self): + expected_acc = 0.7883135059482156 + expected_loss = 0.6474186203172139 + + results = self.run_multiple_choice(model_name="bert-base-cased", task_name="swag", fp16=False) + assert_allclose(results['acc'], expected_acc) + assert_allclose(results['loss'], expected_loss) + + def test_bert_fp16_with_swag(self): + expected_acc = 0.7882135359392183 + expected_loss = 0.6469693916158167 + + results = self.run_multiple_choice(model_name="bert-base-cased", task_name="swag", fp16=True) + assert_allclose(results['acc'], expected_acc) + assert_allclose(results['loss'], expected_loss) + + def run_multiple_choice(self, model_name, task_name, fp16): + model_args = ModelArguments(model_name_or_path=model_name, cache_dir=self.cache_dir) + data_args = DataTrainingArguments(task_name=task_name, data_dir=self.data_dir, + max_seq_length=self.max_seq_length) + + training_args = TrainingArguments(output_dir=os.path.join(self.output_dir, task_name), do_train=True, do_eval=True, + per_gpu_train_batch_size=self.train_batch_size, + per_gpu_eval_batch_size=self.eval_batch_size, + learning_rate=self.learning_rate, num_train_epochs=self.num_train_epochs,local_rank=self.local_rank, + overwrite_output_dir=self.overwrite_output_dir, gradient_accumulation_steps=self.gradient_accumulation_steps, + fp16=fp16, logging_steps=self.logging_steps) + + # Setup logging + logging.basicConfig( + format="%(asctime)s - %(levelname)s - %(name)s - %(message)s", + datefmt="%m/%d/%Y %H:%M:%S", + level=logging.INFO if training_args.local_rank in [-1, 0] else logging.WARN, + ) + logger.warning( + "Process rank: %s, device: %s, n_gpu: %s, distributed training: %s, 16-bits training: %s", + training_args.local_rank, + training_args.device, + training_args.n_gpu, + bool(training_args.local_rank != -1), + training_args.fp16, + ) + logger.info("Training/evaluation parameters %s", training_args) + + set_seed(training_args.seed) + onnxruntime.set_seed(training_args.seed) + + try: + processor = SwagProcessor() + label_list = processor.get_labels() + num_labels = len(label_list) + except KeyError: + raise ValueError("Task not found: %s" % (data_args.task_name)) + + config = AutoConfig.from_pretrained( + model_args.config_name if model_args.config_name else model_args.model_name_or_path, + num_labels=num_labels, + finetuning_task=data_args.task_name, + cache_dir=model_args.cache_dir, + ) + tokenizer = AutoTokenizer.from_pretrained( + model_args.tokenizer_name if model_args.tokenizer_name else model_args.model_name_or_path, + cache_dir=model_args.cache_dir, + ) + + model = AutoModelForMultipleChoice.from_pretrained( + model_args.model_name_or_path, + from_tf=bool(".ckpt" in model_args.model_name_or_path), + config=config, + cache_dir=model_args.cache_dir, + ) + + # Get datasets + train_dataset = ( + MultipleChoiceDataset( + data_dir=data_args.data_dir, + tokenizer=tokenizer, + task=data_args.task_name, + processor=processor, + max_seq_length=data_args.max_seq_length, + overwrite_cache=data_args.overwrite_cache, + mode=Split.train, + ) + if training_args.do_train + else None + ) + eval_dataset = ( + MultipleChoiceDataset( + data_dir=data_args.data_dir, + tokenizer=tokenizer, + task=data_args.task_name, + processor=processor, + max_seq_length=data_args.max_seq_length, + overwrite_cache=data_args.overwrite_cache, + mode=Split.dev, + ) + if training_args.do_eval + else None + ) + + def compute_metrics(p: EvalPrediction) -> Dict: + preds = np.argmax(p.predictions, axis=1) + return {"acc": simple_accuracy(preds, p.label_ids)} + + if model_name.startswith('bert'): + model_desc = ModelDescription([ + IODescription('input_ids', [self.train_batch_size, num_labels, data_args.max_seq_length], torch.int64, num_classes=model.config.vocab_size), + IODescription('attention_mask', [self.train_batch_size, num_labels, data_args.max_seq_length], torch.int64, num_classes=2), + IODescription('token_type_ids', [self.train_batch_size, num_labels, data_args.max_seq_length], torch.int64, num_classes=2), + IODescription('labels', [self.train_batch_size, num_labels], torch.int64, num_classes=num_labels)], [ + IODescription('loss', [], torch.float32), + IODescription('reshaped_logits', [self.train_batch_size, num_labels], torch.float32)]) + else: + model_desc = ModelDescription([ + IODescription('input_ids', ['batch', num_labels, 'max_seq_len_in_batch'], torch.int64, num_classes=model.config.vocab_size), + IODescription('attention_mask', ['batch', num_labels, 'max_seq_len_in_batch'], torch.int64, num_classes=2), + IODescription('labels', ['batch', num_labels], torch.int64, num_classes=num_labels)], [ + IODescription('loss', [], torch.float32), + IODescription('reshaped_logits', ['batch', num_labels], torch.float32)]) + + # Initialize the ORTTrainer within ORTTransformerTrainer + trainer = ORTTransformerTrainer( + model=model, + model_desc=model_desc, + args=training_args, + train_dataset=train_dataset, + eval_dataset=eval_dataset, + compute_metrics=compute_metrics, + ) + + # Training + if training_args.do_train: + trainer.train() + trainer.save_model() + + # Evaluation + results = {} + if training_args.do_eval and training_args.local_rank in [-1, 0]: + logger.info("*** Evaluate ***") + + result = trainer.evaluate() + + logger.info("***** Eval results {} *****".format(data_args.task_name)) + for key, value in result.items(): + logger.info(" %s = %s", key, value) + + results.update(result) + + return results + +if __name__ == "__main__": + unittest.main() diff --git a/orttraining/orttraining/test/python/utils_multiple_choice.py b/orttraining/orttraining/test/python/utils_multiple_choice.py new file mode 100644 index 0000000000..381761998c --- /dev/null +++ b/orttraining/orttraining/test/python/utils_multiple_choice.py @@ -0,0 +1,265 @@ +# adapted from run_multiple_choice.py of huggingface transformers +# https://github.com/huggingface/transformers/blob/master/examples/multiple-choice/utils_multiple_choice.py + +import csv +import glob +import json +import logging +import os +from dataclasses import dataclass +from enum import Enum +from typing import List, Optional + +import tqdm +from filelock import FileLock + +from transformers import PreTrainedTokenizer, is_tf_available, is_torch_available + +import torch +from torch.utils.data.dataset import Dataset + +logger = logging.getLogger(__name__) + + +@dataclass(frozen=True) +class InputExample: + """ + A single training/test example for multiple choice + + Args: + example_id: Unique id for the example. + question: string. The untokenized text of the second sequence (question). + contexts: list of str. The untokenized text of the first sequence (context of corresponding question). + endings: list of str. multiple choice's options. Its length must be equal to contexts' length. + label: (Optional) string. The label of the example. This should be + specified for train and dev examples, but not for test examples. + """ + + example_id: str + question: str + contexts: List[str] + endings: List[str] + label: Optional[str] + + +@dataclass(frozen=True) +class InputFeatures: + """ + A single set of features of data. + Property names are the same names as the corresponding inputs to a model. + """ + + example_id: str + input_ids: List[List[int]] + attention_mask: Optional[List[List[int]]] + token_type_ids: Optional[List[List[int]]] + label: Optional[int] + + +class Split(Enum): + train = "train" + dev = "dev" + test = "test" + +class DataProcessor: + """Base class for data converters for multiple choice data sets.""" + + def get_train_examples(self, data_dir): + """Gets a collection of `InputExample`s for the train set.""" + raise NotImplementedError() + + def get_dev_examples(self, data_dir): + """Gets a collection of `InputExample`s for the dev set.""" + raise NotImplementedError() + + def get_test_examples(self, data_dir): + """Gets a collection of `InputExample`s for the test set.""" + raise NotImplementedError() + + def get_labels(self): + """Gets the list of labels for this data set.""" + raise NotImplementedError() + +class MultipleChoiceDataset(Dataset): + """ + This will be superseded by a framework-agnostic approach + soon. + """ + + features: List[InputFeatures] + + def __init__( + self, + data_dir: str, + tokenizer: PreTrainedTokenizer, + task: str, + processor: DataProcessor, + max_seq_length: Optional[int] = None, + overwrite_cache=False, + mode: Split = Split.train, + ): + processor = processor + + cached_features_file = os.path.join( + data_dir, + "cached_{}_{}_{}_{}".format(mode.value, tokenizer.__class__.__name__, str(max_seq_length), task,), + ) + + # Make sure only the first process in distributed training processes the dataset, + # and the others will use the cache. + lock_path = cached_features_file + ".lock" + with FileLock(lock_path): + + if os.path.exists(cached_features_file) and not overwrite_cache: + logger.info(f"Loading features from cached file {cached_features_file}") + self.features = torch.load(cached_features_file) + else: + logger.info(f"Creating features from dataset file at {data_dir}") + label_list = processor.get_labels() + if mode == Split.dev: + examples = processor.get_dev_examples(data_dir) + elif mode == Split.test: + examples = processor.get_test_examples(data_dir) + else: + examples = processor.get_train_examples(data_dir) + logger.info("Training examples: %s", len(examples)) + # TODO clean up all this to leverage built-in features of tokenizers + self.features = convert_examples_to_features( + examples, + label_list, + max_seq_length, + tokenizer, + pad_on_left=bool(tokenizer.padding_side == "left"), + pad_token=tokenizer.pad_token_id, + pad_token_segment_id=tokenizer.pad_token_type_id, + ) + logger.info("Saving features into cached file %s", cached_features_file) + torch.save(self.features, cached_features_file) + + def __len__(self): + return len(self.features) + + def __getitem__(self, i) -> InputFeatures: + return self.features[i] + +class SwagProcessor(DataProcessor): + """Processor for the SWAG data set.""" + + def get_train_examples(self, data_dir): + """See base class.""" + logger.info("LOOKING AT {} train".format(data_dir)) + return self._create_examples(self._read_csv(os.path.join(data_dir, "train.csv")), "train") + + def get_dev_examples(self, data_dir): + """See base class.""" + logger.info("LOOKING AT {} dev".format(data_dir)) + return self._create_examples(self._read_csv(os.path.join(data_dir, "val.csv")), "dev") + + def get_test_examples(self, data_dir): + """See base class.""" + logger.info("LOOKING AT {} dev".format(data_dir)) + raise ValueError( + "For swag testing, the input file does not contain a label column. It can not be tested in current code" + "setting!" + ) + return self._create_examples(self._read_csv(os.path.join(data_dir, "test.csv")), "test") + + def get_labels(self): + """See base class.""" + return ["0", "1", "2", "3"] + + def _read_csv(self, input_file): + with open(input_file, "r", encoding="utf-8") as f: + return list(csv.reader(f)) + + def _create_examples(self, lines: List[List[str]], type: str): + """Creates examples for the training and dev sets.""" + if type == "train" and lines[0][-1] != "label": + raise ValueError("For training, the input file must contain a label column.") + + examples = [ + InputExample( + example_id=line[2], + question=line[5], # in the swag dataset, the + # common beginning of each + # choice is stored in "sent2". + contexts=[line[4], line[4], line[4], line[4]], + endings=[line[7], line[8], line[9], line[10]], + label=line[11], + ) + for line in lines[1:] # we skip the line with the column names + ] + + return examples + +def convert_examples_to_features( + examples: List[InputExample], + label_list: List[str], + max_length: int, + tokenizer: PreTrainedTokenizer, + pad_token_segment_id=0, + pad_on_left=False, + pad_token=0, + mask_padding_with_zero=True, +) -> List[InputFeatures]: + """ + Loads a data file into a list of `InputFeatures` + """ + + label_map = {label: i for i, label in enumerate(label_list)} + + features = [] + for (ex_index, example) in tqdm.tqdm(enumerate(examples), desc="convert examples to features"): + if ex_index % 10000 == 0: + logger.info("Writing example %d of %d" % (ex_index, len(examples))) + choices_inputs = [] + for ending_idx, (context, ending) in enumerate(zip(example.contexts, example.endings)): + text_a = context + if example.question.find("_") != -1: + # this is for cloze question + text_b = example.question.replace("_", ending) + else: + text_b = example.question + " " + ending + + inputs = tokenizer.encode_plus( + text_a, + text_b, + add_special_tokens=True, + max_length=max_length, + pad_to_max_length=True, + return_overflowing_tokens=True, + ) + if "num_truncated_tokens" in inputs and inputs["num_truncated_tokens"] > 0: + logger.info( + "Attention! you are cropping tokens (swag task is ok). " + "If you are training ARC and RACE and you are poping question + options," + "you need to try to use a bigger max seq length!" + ) + + choices_inputs.append(inputs) + + label = label_map[example.label] + + input_ids = [x["input_ids"] for x in choices_inputs] + attention_mask = ( + [x["attention_mask"] for x in choices_inputs] if "attention_mask" in choices_inputs[0] else None + ) + token_type_ids = ( + [x["token_type_ids"] for x in choices_inputs] if "token_type_ids" in choices_inputs[0] else None + ) + + features.append( + InputFeatures( + example_id=example.example_id, + input_ids=input_ids, + attention_mask=attention_mask, + token_type_ids=token_type_ids, + label=label, + ) + ) + + for f in features[:2]: + logger.info("*** Example ***") + logger.info("feature: %s" % f) + + return features diff --git a/tools/ci_build/build.py b/tools/ci_build/build.py index a1f8962c66..9462795096 100755 --- a/tools/ci_build/build.py +++ b/tools/ci_build/build.py @@ -1137,6 +1137,10 @@ def run_training_python_frontend_e2e_tests(cwd): [sys.executable, 'orttraining_run_glue.py', 'ORTGlueTest.test_roberta_fp16_with_mrpc', '-v'], cwd=cwd, env={'CUDA_VISIBLE_DEVICES': '0'}) + run_subprocess( + [sys.executable, 'orttraining_run_multiple_choice.py', 'ORTMultipleChoiceTest.test_bert_fp16_with_swag', '-v'], + cwd=cwd, env={'CUDA_VISIBLE_DEVICES': '0'}) + run_subprocess([sys.executable, 'onnxruntime_test_ort_trainer_with_mixed_precision.py'], cwd=cwd) run_subprocess([ diff --git a/tools/ci_build/github/azure-pipelines/orttraining-linux-gpu-frontend-test-ci-pipeline.yml b/tools/ci_build/github/azure-pipelines/orttraining-linux-gpu-frontend-test-ci-pipeline.yml index 37562df5d8..827a930699 100644 --- a/tools/ci_build/github/azure-pipelines/orttraining-linux-gpu-frontend-test-ci-pipeline.yml +++ b/tools/ci_build/github/azure-pipelines/orttraining-linux-gpu-frontend-test-ci-pipeline.yml @@ -6,14 +6,25 @@ trigger: none jobs: - job: Onnxruntime_Linux_GPU_Training_FrontEnd - timeoutInMinutes: 120 + timeoutInMinutes: 240 steps: - checkout: self clean: true submodules: recursive - - template: templates/linux-set-variables-and-download.yml + - template: templates/set-test-data-variables-step.yml + + - task: CmdLine@2 + displayName: 'Clean untagged docker images' + inputs: + script: | + docker rm $(docker ps -a | grep Exited | awk '{print $1;}') || true + docker container prune -f + docker image prune -f + workingDirectory: $(Build.BinariesDirectory) + continueOnError: true + condition: always() # insert a python frontend test data preparation step here