[s2s testing] turn all to unittests, use auto-delete temp dirs (#7859)

2020-10-17 11:33:21 -07:00
parent dc552b9b70
commit 9f7b2b2432
6 changed files with 635 additions and 655 deletions
--- a/examples/seq2seq/test_seq2seq_examples.py
+++ b/examples/seq2seq/test_seq2seq_examples.py
@@ -3,7 +3,6 @@ import logging
 import os
 import sys
 import tempfile
-import unittest
 from pathlib import Path
 from unittest.mock import patch

@@ -15,11 +14,12 @@ import lightning_base
 from convert_pl_checkpoint_to_hf import convert_pl_to_hf
 from distillation import distill_main
 from finetune import SummarizationModule, main
+from parameterized import parameterized
 from run_eval import generate_summaries_or_translations, run_generate
 from run_eval_search import run_search
 from transformers import AutoConfig, AutoModelForSeq2SeqLM
 from transformers.hf_api import HfApi
-from transformers.testing_utils import CaptureStderr, CaptureStdout, require_multigpu, require_torch_and_cuda, slow
+from transformers.testing_utils import CaptureStderr, CaptureStdout, TestCasePlus, require_torch_and_cuda, slow
 from utils import ROUGE_KEYS, label_smoothed_nll_loss, lmap, load_json


@@ -52,7 +52,7 @@ CHEAP_ARGS = {
    "student_decoder_layers": 1,
    "val_check_interval": 1.0,
    "output_dir": "",
-    "fp16": False,  # TODO: set this to CUDA_AVAILABLE if ci installs apex or start using native amp
+    "fp16": False,  # TODO(SS): set this to CUDA_AVAILABLE if ci installs apex or start using native amp
    "no_teacher": False,
    "fp16_opt_level": "O1",
    "gpus": 1 if CUDA_AVAILABLE else 0,
@@ -88,6 +88,7 @@ CHEAP_ARGS = {
    "student_encoder_layers": 1,
    "freeze_encoder": False,
    "auto_scale_batch_size": False,
+    "overwrite_output_dir": False,
 }


@@ -110,15 +111,14 @@ logger.addHandler(stream_handler)
 logging.disable(logging.CRITICAL)  # remove noisy download output from tracebacks


-def make_test_data_dir(**kwargs):
-    tmp_dir = Path(tempfile.mkdtemp(**kwargs))
+def make_test_data_dir(tmp_dir):
    for split in ["train", "val", "test"]:
-        _dump_articles((tmp_dir / f"{split}.source"), ARTICLES)
-        _dump_articles((tmp_dir / f"{split}.target"), SUMMARIES)
+        _dump_articles(os.path.join(tmp_dir, f"{split}.source"), ARTICLES)
+        _dump_articles(os.path.join(tmp_dir, f"{split}.target"), SUMMARIES)
    return tmp_dir


-class TestSummarizationDistiller(unittest.TestCase):
+class TestSummarizationDistiller(TestCasePlus):
    @classmethod
    def setUpClass(cls):
        logging.disable(logging.CRITICAL)  # remove noisy download output from tracebacks
@@ -143,17 +143,6 @@ class TestSummarizationDistiller(unittest.TestCase):
                failures.append(m)
        assert not failures, f"The following models could not be loaded through AutoConfig: {failures}"

-    @require_multigpu
-    @unittest.skip("Broken at the moment")
-    def test_multigpu(self):
-        updates = dict(
-            no_teacher=True,
-            freeze_encoder=True,
-            gpus=2,
-            sortish_sampler=True,
-        )
-        self._test_distiller_cli(updates, check_contents=False)
-
    def test_distill_no_teacher(self):
        updates = dict(student_encoder_layers=2, student_decoder_layers=1, no_teacher=True)
        self._test_distiller_cli(updates)
@@ -173,12 +162,12 @@ class TestSummarizationDistiller(unittest.TestCase):
        self.assertEqual(1, len(ckpts))
        transformer_ckpts = list(Path(model.output_dir).glob("**/*.bin"))
        self.assertEqual(len(transformer_ckpts), 2)
-        examples = lmap(str.strip, model.hparams.data_dir.joinpath("test.source").open().readlines())
-        out_path = tempfile.mktemp()
+        examples = lmap(str.strip, Path(model.hparams.data_dir).joinpath("test.source").open().readlines())
+        out_path = tempfile.mktemp()  # XXX: not being cleaned up
        generate_summaries_or_translations(examples, out_path, str(model.output_dir / "best_tfmr"))
        self.assertTrue(Path(out_path).exists())

-        out_path_new = tempfile.mkdtemp()
+        out_path_new = self.get_auto_remove_tmp_dir()
        convert_pl_to_hf(ckpts[0], transformer_ckpts[0].parent, out_path_new)
        assert os.path.exists(os.path.join(out_path_new, "pytorch_model.bin"))

@@ -253,8 +242,8 @@ class TestSummarizationDistiller(unittest.TestCase):
        )
        default_updates.update(updates)
        args_d: dict = CHEAP_ARGS.copy()
-        tmp_dir = make_test_data_dir()
-        output_dir = tempfile.mkdtemp(prefix="output_")
+        tmp_dir = make_test_data_dir(tmp_dir=self.get_auto_remove_tmp_dir())
+        output_dir = self.get_auto_remove_tmp_dir()

        args_d.update(data_dir=tmp_dir, output_dir=output_dir, **default_updates)
        model = distill_main(argparse.Namespace(**args_d))
@@ -279,256 +268,253 @@ class TestSummarizationDistiller(unittest.TestCase):
        return model


-def run_eval_tester(model):
-    input_file_name = Path(tempfile.mkdtemp()) / "utest_input.source"
-    output_file_name = input_file_name.parent / "utest_output.txt"
-    assert not output_file_name.exists()
-    articles = [" New York (CNN)When Liana Barrientos was 23 years old, she got married in Westchester County."]
-    _dump_articles(input_file_name, articles)
-    score_path = str(Path(tempfile.mkdtemp()) / "scores.json")
-    task = "translation_en_to_de" if model == T5_TINY else "summarization"
-    testargs = f"""
-        run_eval_search.py
-        {model}
-        {input_file_name}
-        {output_file_name}
-        --score_path {score_path}
-        --task {task}
-        --num_beams 2
-        --length_penalty 2.0
-        """.split()
+class TestTheRest(TestCasePlus):
+    def run_eval_tester(self, model):
+        input_file_name = Path(self.get_auto_remove_tmp_dir()) / "utest_input.source"
+        output_file_name = input_file_name.parent / "utest_output.txt"
+        assert not output_file_name.exists()
+        articles = [" New York (CNN)When Liana Barrientos was 23 years old, she got married in Westchester County."]
+        _dump_articles(input_file_name, articles)

-    with patch.object(sys, "argv", testargs):
-        run_generate()
-        assert Path(output_file_name).exists()
-        os.remove(Path(output_file_name))
+        score_path = str(Path(self.get_auto_remove_tmp_dir()) / "scores.json")
+        task = "translation_en_to_de" if model == T5_TINY else "summarization"
+        testargs = f"""
+            run_eval_search.py
+            {model}
+            {input_file_name}
+            {output_file_name}
+            --score_path {score_path}
+            --task {task}
+            --num_beams 2
+            --length_penalty 2.0
+            """.split()

+        with patch.object(sys, "argv", testargs):
+            run_generate()
+            assert Path(output_file_name).exists()
+            # os.remove(Path(output_file_name))

-# test one model to quickly (no-@slow) catch simple problems and do an
-# extensive testing of functionality with multiple models as @slow separately
-def test_run_eval():
-    run_eval_tester(T5_TINY)
+    # test one model to quickly (no-@slow) catch simple problems and do an
+    # extensive testing of functionality with multiple models as @slow separately
+    def test_run_eval(self):
+        self.run_eval_tester(T5_TINY)

+    # any extra models should go into the list here - can be slow
+    @parameterized.expand([BART_TINY, MBART_TINY])
+    @slow
+    def test_run_eval_slow(self, model):
+        self.run_eval_tester(model)

-# any extra models should go into the list here - can be slow
-@slow
-@pytest.mark.parametrize("model", [BART_TINY, MBART_TINY])
-def test_run_eval_slow(model):
-    run_eval_tester(model)
+    # testing with 2 models to validate: 1. translation (t5) 2. summarization (mbart)
+    @parameterized.expand([T5_TINY, MBART_TINY])
+    @slow
+    def test_run_eval_search(self, model):
+        input_file_name = Path(self.get_auto_remove_tmp_dir()) / "utest_input.source"
+        output_file_name = input_file_name.parent / "utest_output.txt"
+        assert not output_file_name.exists()

+        text = {
+            "en": ["Machine learning is great, isn't it?", "I like to eat bananas", "Tomorrow is another great day!"],
+            "de": [
+                "Maschinelles Lernen ist großartig, oder?",
+                "Ich esse gerne Bananen",
+                "Morgen ist wieder ein toller Tag!",
+            ],
+        }

-# testing with 2 models to validate: 1. translation (t5) 2. summarization (mbart)
-@slow
-@pytest.mark.parametrize("model", [T5_TINY, MBART_TINY])
-def test_run_eval_search(model):
-    input_file_name = Path(tempfile.mkdtemp()) / "utest_input.source"
-    output_file_name = input_file_name.parent / "utest_output.txt"
-    assert not output_file_name.exists()
+        tmp_dir = Path(self.get_auto_remove_tmp_dir())
+        score_path = str(tmp_dir / "scores.json")
+        reference_path = str(tmp_dir / "val.target")
+        _dump_articles(input_file_name, text["en"])
+        _dump_articles(reference_path, text["de"])
+        task = "translation_en_to_de" if model == T5_TINY else "summarization"
+        testargs = f"""
+            run_eval_search.py
+            {model}
+            {str(input_file_name)}
+            {str(output_file_name)}
+            --score_path {score_path}
+            --reference_path {reference_path}
+            --task {task}
+            """.split()
+        testargs.extend(["--search", "num_beams=1:2 length_penalty=0.9:1.0"])

-    text = {
-        "en": ["Machine learning is great, isn't it?", "I like to eat bananas", "Tomorrow is another great day!"],
-        "de": [
-            "Maschinelles Lernen ist großartig, oder?",
-            "Ich esse gerne Bananen",
-            "Morgen ist wieder ein toller Tag!",
-        ],
-    }
+        with patch.object(sys, "argv", testargs):
+            with CaptureStdout() as cs:
+                run_search()
+            expected_strings = [" num_beams | length_penalty", model, "Best score args"]
+            un_expected_strings = ["Info"]
+            if "translation" in task:
+                expected_strings.append("bleu")
+            else:
+                expected_strings.extend(ROUGE_KEYS)
+            for w in expected_strings:
+                assert w in cs.out
+            for w in un_expected_strings:
+                assert w not in cs.out
+            assert Path(output_file_name).exists()
+            os.remove(Path(output_file_name))

-    tmp_dir = Path(tempfile.mkdtemp())
-    score_path = str(tmp_dir / "scores.json")
-    reference_path = str(tmp_dir / "val.target")
-    _dump_articles(input_file_name, text["en"])
-    _dump_articles(reference_path, text["de"])
-    task = "translation_en_to_de" if model == T5_TINY else "summarization"
-    testargs = f"""
-        run_eval_search.py
-        {model}
-        {str(input_file_name)}
-        {str(output_file_name)}
-        --score_path {score_path}
-        --reference_path {reference_path}
-        --task {task}
-        """.split()
-    testargs.extend(["--search", "num_beams=1:2 length_penalty=0.9:1.0"])
+    @parameterized.expand(
+        [T5_TINY, BART_TINY, MBART_TINY, MARIAN_TINY, FSMT_TINY],
+    )
+    def test_finetune(self, model):
+        args_d: dict = CHEAP_ARGS.copy()
+        task = "translation" if model in [MBART_TINY, MARIAN_TINY, FSMT_TINY] else "summarization"
+        args_d["label_smoothing"] = 0.1 if task == "translation" else 0

-    with patch.object(sys, "argv", testargs):
-        with CaptureStdout() as cs:
-            run_search()
-        expected_strings = [" num_beams | length_penalty", model, "Best score args"]
-        un_expected_strings = ["Info"]
-        if "translation" in task:
-            expected_strings.append("bleu")
+        tmp_dir = make_test_data_dir(tmp_dir=self.get_auto_remove_tmp_dir())
+        output_dir = self.get_auto_remove_tmp_dir()
+        args_d.update(
+            data_dir=tmp_dir,
+            model_name_or_path=model,
+            tokenizer_name=None,
+            train_batch_size=2,
+            eval_batch_size=2,
+            output_dir=output_dir,
+            do_predict=True,
+            task=task,
+            src_lang="en_XX",
+            tgt_lang="ro_RO",
+            freeze_encoder=True,
+            freeze_embeds=True,
+        )
+        assert "n_train" in args_d
+        args = argparse.Namespace(**args_d)
+        module = main(args)
+
+        input_embeds = module.model.get_input_embeddings()
+        assert not input_embeds.weight.requires_grad
+        if model == T5_TINY:
+            lm_head = module.model.lm_head
+            assert not lm_head.weight.requires_grad
+            assert (lm_head.weight == input_embeds.weight).all().item()
+        elif model == FSMT_TINY:
+            fsmt = module.model.model
+            embed_pos = fsmt.decoder.embed_positions
+            assert not embed_pos.weight.requires_grad
+            assert not fsmt.decoder.embed_tokens.weight.requires_grad
+            # check that embeds are not the same
+            assert fsmt.decoder.embed_tokens != fsmt.encoder.embed_tokens
        else:
-            expected_strings.extend(ROUGE_KEYS)
-        for w in expected_strings:
-            assert w in cs.out
-        for w in un_expected_strings:
-            assert w not in cs.out
-        assert Path(output_file_name).exists()
-        os.remove(Path(output_file_name))
+            bart = module.model.model
+            embed_pos = bart.decoder.embed_positions
+            assert not embed_pos.weight.requires_grad
+            assert not bart.shared.weight.requires_grad
+            # check that embeds are the same
+            assert bart.decoder.embed_tokens == bart.encoder.embed_tokens
+            assert bart.decoder.embed_tokens == bart.shared

+        example_batch = load_json(module.output_dir / "text_batch.json")
+        assert isinstance(example_batch, dict)
+        assert len(example_batch) >= 4

-@pytest.mark.parametrize(
-    "model",
-    [T5_TINY, BART_TINY, MBART_TINY, MARIAN_TINY, FSMT_TINY],
-)
-def test_finetune(model):
-    args_d: dict = CHEAP_ARGS.copy()
-    task = "translation" if model in [MBART_TINY, MARIAN_TINY, FSMT_TINY] else "summarization"
-    args_d["label_smoothing"] = 0.1 if task == "translation" else 0
+    def test_finetune_extra_model_args(self):
+        args_d: dict = CHEAP_ARGS.copy()

-    tmp_dir = make_test_data_dir()
-    output_dir = tempfile.mkdtemp(prefix="output_")
-    args_d.update(
-        data_dir=tmp_dir,
-        model_name_or_path=model,
-        tokenizer_name=None,
-        train_batch_size=2,
-        eval_batch_size=2,
-        output_dir=output_dir,
-        do_predict=True,
-        task=task,
-        src_lang="en_XX",
-        tgt_lang="ro_RO",
-        freeze_encoder=True,
-        freeze_embeds=True,
-    )
-    assert "n_train" in args_d
-    args = argparse.Namespace(**args_d)
-    module = main(args)
+        task = "summarization"
+        tmp_dir = make_test_data_dir(tmp_dir=self.get_auto_remove_tmp_dir())

-    input_embeds = module.model.get_input_embeddings()
-    assert not input_embeds.weight.requires_grad
-    if model == T5_TINY:
-        lm_head = module.model.lm_head
-        assert not lm_head.weight.requires_grad
-        assert (lm_head.weight == input_embeds.weight).all().item()
-    elif model == FSMT_TINY:
-        fsmt = module.model.model
-        embed_pos = fsmt.decoder.embed_positions
-        assert not embed_pos.weight.requires_grad
-        assert not fsmt.decoder.embed_tokens.weight.requires_grad
-        # check that embeds are not the same
-        assert fsmt.decoder.embed_tokens != fsmt.encoder.embed_tokens
-    else:
-        bart = module.model.model
-        embed_pos = bart.decoder.embed_positions
-        assert not embed_pos.weight.requires_grad
-        assert not bart.shared.weight.requires_grad
-        # check that embeds are the same
-        assert bart.decoder.embed_tokens == bart.encoder.embed_tokens
-        assert bart.decoder.embed_tokens == bart.shared
+        args_d.update(
+            data_dir=tmp_dir,
+            tokenizer_name=None,
+            train_batch_size=2,
+            eval_batch_size=2,
+            do_predict=False,
+            task=task,
+            src_lang="en_XX",
+            tgt_lang="ro_RO",
+            freeze_encoder=True,
+            freeze_embeds=True,
+        )

-    example_batch = load_json(module.output_dir / "text_batch.json")
-    assert isinstance(example_batch, dict)
-    assert len(example_batch) >= 4
-
-
-def test_finetune_extra_model_args():
-    args_d: dict = CHEAP_ARGS.copy()
-
-    task = "summarization"
-    tmp_dir = make_test_data_dir()
-
-    args_d.update(
-        data_dir=tmp_dir,
-        tokenizer_name=None,
-        train_batch_size=2,
-        eval_batch_size=2,
-        do_predict=False,
-        task=task,
-        src_lang="en_XX",
-        tgt_lang="ro_RO",
-        freeze_encoder=True,
-        freeze_embeds=True,
-    )
-
-    # test models whose config includes the extra_model_args
-    model = BART_TINY
-    output_dir = tempfile.mkdtemp(prefix="output_1_")
-    args_d1 = args_d.copy()
-    args_d1.update(
-        model_name_or_path=model,
-        output_dir=output_dir,
-    )
-    extra_model_params = ("encoder_layerdrop", "decoder_layerdrop", "dropout", "attention_dropout")
-    for p in extra_model_params:
-        args_d1[p] = 0.5
-    args = argparse.Namespace(**args_d1)
-    model = main(args)
-    for p in extra_model_params:
-        assert getattr(model.config, p) == 0.5, f"failed to override the model config for param {p}"
-
-    # test models whose config doesn't include the extra_model_args
-    model = T5_TINY
-    output_dir = tempfile.mkdtemp(prefix="output_2_")
-    args_d2 = args_d.copy()
-    args_d2.update(
-        model_name_or_path=model,
-        output_dir=output_dir,
-    )
-    unsupported_param = "encoder_layerdrop"
-    args_d2[unsupported_param] = 0.5
-    args = argparse.Namespace(**args_d2)
-    with pytest.raises(Exception) as excinfo:
+        # test models whose config includes the extra_model_args
+        model = BART_TINY
+        output_dir = self.get_auto_remove_tmp_dir()
+        args_d1 = args_d.copy()
+        args_d1.update(
+            model_name_or_path=model,
+            output_dir=output_dir,
+        )
+        extra_model_params = ("encoder_layerdrop", "decoder_layerdrop", "dropout", "attention_dropout")
+        for p in extra_model_params:
+            args_d1[p] = 0.5
+        args = argparse.Namespace(**args_d1)
        model = main(args)
-    assert str(excinfo.value) == f"model config doesn't have a `{unsupported_param}` attribute"
+        for p in extra_model_params:
+            assert getattr(model.config, p) == 0.5, f"failed to override the model config for param {p}"

+        # test models whose config doesn't include the extra_model_args
+        model = T5_TINY
+        output_dir = self.get_auto_remove_tmp_dir()
+        args_d2 = args_d.copy()
+        args_d2.update(
+            model_name_or_path=model,
+            output_dir=output_dir,
+        )
+        unsupported_param = "encoder_layerdrop"
+        args_d2[unsupported_param] = 0.5
+        args = argparse.Namespace(**args_d2)
+        with pytest.raises(Exception) as excinfo:
+            model = main(args)
+        assert str(excinfo.value) == f"model config doesn't have a `{unsupported_param}` attribute"

-def test_finetune_lr_schedulers():
-    args_d: dict = CHEAP_ARGS.copy()
+    def test_finetune_lr_schedulers(self):
+        args_d: dict = CHEAP_ARGS.copy()

-    task = "summarization"
-    tmp_dir = make_test_data_dir()
+        task = "summarization"
+        tmp_dir = make_test_data_dir(tmp_dir=self.get_auto_remove_tmp_dir())

-    model = BART_TINY
-    output_dir = tempfile.mkdtemp(prefix="output_1_")
+        model = BART_TINY
+        output_dir = self.get_auto_remove_tmp_dir()

-    args_d.update(
-        data_dir=tmp_dir,
-        model_name_or_path=model,
-        output_dir=output_dir,
-        tokenizer_name=None,
-        train_batch_size=2,
-        eval_batch_size=2,
-        do_predict=False,
-        task=task,
-        src_lang="en_XX",
-        tgt_lang="ro_RO",
-        freeze_encoder=True,
-        freeze_embeds=True,
-    )
+        args_d.update(
+            data_dir=tmp_dir,
+            model_name_or_path=model,
+            output_dir=output_dir,
+            tokenizer_name=None,
+            train_batch_size=2,
+            eval_batch_size=2,
+            do_predict=False,
+            task=task,
+            src_lang="en_XX",
+            tgt_lang="ro_RO",
+            freeze_encoder=True,
+            freeze_embeds=True,
+        )

-    # emulate finetune.py
-    parser = argparse.ArgumentParser()
-    parser = pl.Trainer.add_argparse_args(parser)
-    parser = SummarizationModule.add_model_specific_args(parser, os.getcwd())
-    args = {"--help": True}
+        # emulate finetune.py
+        parser = argparse.ArgumentParser()
+        parser = pl.Trainer.add_argparse_args(parser)
+        parser = SummarizationModule.add_model_specific_args(parser, os.getcwd())
+        args = {"--help": True}

-    # --help test
-    with pytest.raises(SystemExit) as excinfo:
-        with CaptureStdout() as cs:
-            args = parser.parse_args(args)
-        assert False, "--help is expected to sys.exit"
-    assert excinfo.type == SystemExit
-    expected = lightning_base.arg_to_scheduler_metavar
-    assert expected in cs.out, "--help is expected to list the supported schedulers"
+        # --help test
+        with pytest.raises(SystemExit) as excinfo:
+            with CaptureStdout() as cs:
+                args = parser.parse_args(args)
+            assert False, "--help is expected to sys.exit"
+        assert excinfo.type == SystemExit
+        expected = lightning_base.arg_to_scheduler_metavar
+        assert expected in cs.out, "--help is expected to list the supported schedulers"

-    # --lr_scheduler=non_existing_scheduler test
-    unsupported_param = "non_existing_scheduler"
-    args = {f"--lr_scheduler={unsupported_param}"}
-    with pytest.raises(SystemExit) as excinfo:
-        with CaptureStderr() as cs:
-            args = parser.parse_args(args)
-        assert False, "invalid argument is expected to sys.exit"
-    assert excinfo.type == SystemExit
-    expected = f"invalid choice: '{unsupported_param}'"
-    assert expected in cs.err, f"should have bailed on invalid choice of scheduler {unsupported_param}"
+        # --lr_scheduler=non_existing_scheduler test
+        unsupported_param = "non_existing_scheduler"
+        args = {f"--lr_scheduler={unsupported_param}"}
+        with pytest.raises(SystemExit) as excinfo:
+            with CaptureStderr() as cs:
+                args = parser.parse_args(args)
+            assert False, "invalid argument is expected to sys.exit"
+        assert excinfo.type == SystemExit
+        expected = f"invalid choice: '{unsupported_param}'"
+        assert expected in cs.err, f"should have bailed on invalid choice of scheduler {unsupported_param}"

-    # --lr_scheduler=existing_scheduler test
-    supported_param = "cosine"
-    args_d1 = args_d.copy()
-    args_d1["lr_scheduler"] = supported_param
-    args = argparse.Namespace(**args_d1)
-    model = main(args)
-    assert getattr(model.hparams, "lr_scheduler") == supported_param, f"lr_scheduler={supported_param} shouldn't fail"
+        # --lr_scheduler=existing_scheduler test
+        supported_param = "cosine"
+        args_d1 = args_d.copy()
+        args_d1["lr_scheduler"] = supported_param
+        args = argparse.Namespace(**args_d1)
+        model = main(args)
+        assert (
+            getattr(model.hparams, "lr_scheduler") == supported_param
+        ), f"lr_scheduler={supported_param} shouldn't fail"