From 03fa47a2ff19eecbc72170f9747539a1eac7f0fc Mon Sep 17 00:00:00 2001 From: Maxim Vafin Date: Tue, 16 Apr 2024 16:27:34 +0200 Subject: [PATCH] Refactor hf GHA tests (#24048) ### Details: - *Refactor hf tests* ### Tickets: - *ticket-id* --- .../pytorch/hf_transformers_models | 18 +- .../pytorch/test_hf_transformers.py | 185 ++++-------------- 2 files changed, 45 insertions(+), 158 deletions(-) diff --git a/tests/model_hub_tests/pytorch/hf_transformers_models b/tests/model_hub_tests/pytorch/hf_transformers_models index 775797f1b2c..bf76bc90e5a 100644 --- a/tests/model_hub_tests/pytorch/hf_transformers_models +++ b/tests/model_hub_tests/pytorch/hf_transformers_models @@ -23,7 +23,7 @@ asapp/sew-d-base-plus-400k-ft-ls100h,sew-d ashishpatel26/span-marker-bert-base-fewnerd-coarse-super,span-marker,skip,Load problem asi/albert-act-tiny,albert_act,skip,Load problem BAAI/AltCLIP,altclip -BAAI/AquilaCode-py,aquila,skip,Load problem +BAAI/AquilaCode-py,aquila bana513/opennmt-translator-en-hu,opennmt-translator,skip,Load problem benjamin/wtp-bert-mini,bert-char,skip,Load problem benjamin/wtp-canine-s-1l,la-canine,skip,Load problem @@ -79,6 +79,7 @@ facebook/m2m100_418M,m2m_100 facebook/mask2former-swin-base-coco-panoptic,mask2former facebook/maskformer-swin-base-coco,maskformer facebook/mbart-large-50-many-to-many-mmt,mbart +facebook/mms-lid-126,wav2vec2 facebook/mms-tts-eng,vits,xfail,Accuracy failed: results cannot be broadcasted facebook/musicgen-small,musicgen facebook/opt-125m,opt @@ -90,7 +91,7 @@ facebook/wmt19-ru-en,fsmt,xfail,Tracing problem facebook/xglm-7.5B,xglm facebook/xlm-roberta-xl,xlm-roberta-xl facebook/xmod-base,xmod -flax-community/ft5-cnn-dm,f_t5,skip,Load problem +flax-community/ft5-cnn-dm,f_t5 fnlp/elasticbert-base,elasticbert,skip,Load problem FranzStrauss/ponet-base-uncased,ponet,skip,Load problem funnel-transformer/small,funnel @@ -117,7 +118,7 @@ google/reformer-crime-and-punishment,reformer,xfail,Tracing problem google/tapas-large-finetuned-wtq,tapas google/vit-hybrid-base-bit-384,vit-hybrid,skip,Load problem google/vivit-b-16x2-kinetics400,vivit -Goutham-Vignesh/ContributionSentClassification-scibert,scibert,skip,Load problem +Goutham-Vignesh/ContributionSentClassification-scibert,scibert gpt2,gpt2 Graphcore/groupbert-base-uncased,groupbert,skip,Load problem haoranzhao419/saffu-100M-0.1,saffu-100M-0.1,skip,Load problem @@ -166,7 +167,7 @@ HJHGJGHHG/GAU-Base-Full,gau,skip,Load problem huggingface/autoformer-tourism-monthly,autoformer,skip,Load problem huggingface/informer-tourism-monthly,informer,skip,Load problem huggingface/time-series-transformer-tourism-monthly,time_series_transformer,skip,Load problem -HuggingFaceM4/tiny-random-idefics,idefics,xfail,tracing error: Please check correctness of provided example_input (eval was correct but trace failed with incommatible tuples and tensors) +HuggingFaceM4/tiny-random-idefics,idefics,xfail,Unsupported op aten::any aten::einsum prim::TupleConstruct prim::TupleUnpack HuggingFaceM4/tiny-random-vllama-clip,vllama,skip,Load problem HuggingFaceM4/tiny-random-vopt-clip,vopt,skip,Load problem HuiHuang/gpt3-damo-base-zh,gpt3,skip,Load problem @@ -184,7 +185,6 @@ jambran/depression-classification,DepressionDetection,skip,Load problem Jellywibble/dalio-reward-charlie-v1,reward-model,skip,Load problem JonasGeiping/crammed-bert-legacy,crammedBERT,skip,Load problem jonatasgrosman/wav2vec2-large-xlsr-53-english,wav2vec2,xfail,Unsupported op aten::index_put_ prim::TupleConstruct -facebook/mms-lid-126,wav2vec2 Joqsan/test-my-fnet,my_fnet,skip,Load problem jozhang97/deta-swin-large,deta,skip,Load problem jploski/retnet-mini-shakespeare,retnet,skip,Load problem @@ -247,7 +247,7 @@ microsoft/markuplm-base,markuplm microsoft/prophetnet-large-uncased-squad-qg,prophetnet microsoft/resnet-50,resnet microsoft/speecht5_hifigan,hifigan,skip,Load problem -microsoft/speecht5_tts,speecht5,xfail,Tracing error: hangs with no error (probably because of infinite while inside generate) +microsoft/speecht5_tts,speecht5,xfail,Unsupported op aten::bernoulli microsoft/swinv2-tiny-patch4-window8-256,swinv2 microsoft/table-transformer-detection,table-transformer microsoft/unispeech-1350-en-17h-ky-ft-1h,unispeech @@ -303,7 +303,7 @@ paulhindemith/test-zeroshot,test-zeroshot,skip,Load problem PGT/orig-nystromformer-s-artificial-balanced-max500-490000-0,graph_nystromformer,skip,Load problem pie/example-ner-spanclf-conll03,TransformerSpanClassificationModel,skip,Load problem pie/example-re-textclf-tacred,TransformerTextClassificationModel,skip,Load problem -pleisto/yuren-baichuan-7b,multimodal_llama,skip,Load problem +pleisto/yuren-baichuan-7b,multimodal_llama predictia/europe_reanalysis_downscaler_convbaseline,convbilinear,skip,Load problem predictia/europe_reanalysis_downscaler_convswin2sr,conv_swin2sr,skip,Load problem pszemraj/led-large-book-summary,led @@ -341,7 +341,7 @@ shikhartuli/flexibert-mini,flexibert,skip,Load problem shikras/shikra-7b-delta-v1-0708,shikra,skip,Load problem shi-labs/dinat-mini-in1k-224,dinat,xfail,Accuracy validation failed shi-labs/nat-mini-in1k-224,nat,xfail,Accuracy validation failed -shi-labs/oneformer_ade20k_swin_large,oneformer,xfail,Tracing error: Please check correctness of provided example_input (but eval was correct) +shi-labs/oneformer_ade20k_swin_large,oneformer,xfail,Different number of outputs between framework and OpenVINO shuqi/seed-encoder,seed_encoder,skip,Load problem sijunhe/nezha-cn-base,nezha sjiang1/codecse,roberta_for_cl,skip,Load problem @@ -352,7 +352,7 @@ solotimes/lavibe_base,donut,skip,Load problem songlab/gpn-brassicales,ConvNet,skip,Load problem speechbrain/m-ctc-t-large,mctct Splend1dchan/wav2vec2-large-lv60_t5lephone-small_lna_bs64,speechmix,skip,Load problem -stefan-it/bort-full,bort,skip,Load problem +stefan-it/bort-full,bort SteveZhan/my-resnet50d,resnet_steve,skip,Load problem suno/bark,bark,skip,Load problem surajnair/r3m-50,r3m,skip,Load problem diff --git a/tests/model_hub_tests/pytorch/test_hf_transformers.py b/tests/model_hub_tests/pytorch/test_hf_transformers.py index 94759c05265..bd27638159b 100644 --- a/tests/model_hub_tests/pytorch/test_hf_transformers.py +++ b/tests/model_hub_tests/pytorch/test_hf_transformers.py @@ -8,9 +8,12 @@ import torch from huggingface_hub import model_info from models_hub_common.constants import hf_hub_cache_dir from models_hub_common.utils import cleanup_dir +import transformers +from transformers import AutoConfig, AutoModel, AutoProcessor, AutoTokenizer, AutoFeatureExtractor, AutoModelForTextToWaveform, \ + CLIPFeatureExtractor, XCLIPVisionModel, T5Tokenizer, VisionEncoderDecoderModel, ViTImageProcessor, BlipProcessor, BlipForConditionalGeneration, \ + SpeechT5Processor, SpeechT5ForTextToSpeech, LayoutLMv2Processor, Pix2StructForConditionalGeneration, RetriBertTokenizer, VivitImageProcessor -from torch_utils import TestTorchConvertModel -from torch_utils import process_pytest_marks +from torch_utils import TestTorchConvertModel, process_pytest_marks def is_gptq_model(config): config_dict = config.to_dict() if not isinstance(config, dict) else config @@ -92,16 +95,14 @@ class TestTransformersModel(TestTorchConvertModel): from PIL import Image import requests - self.infer_timeout = 1000 + self.infer_timeout = 1800 url = "http://images.cocodataset.org/val2017/000000039769.jpg" self.image = Image.open(requests.get(url, stream=True).raw) self.cuda_available, self.gptq_postinit = None, None def load_model(self, name, type): - import torch name_suffix = '' - from transformers import AutoConfig if name.find(':') != -1: name_suffix = name[name.find(':') + 1:] name = name[:name.find(':')] @@ -128,73 +129,45 @@ class TestTransformersModel(TestTorchConvertModel): except: auto_model = None if "clip_vision_model" in mi.tags: - from transformers import CLIPVisionModel, CLIPFeatureExtractor - model = CLIPVisionModel.from_pretrained(name, torchscript=True) preprocessor = CLIPFeatureExtractor.from_pretrained(name) encoded_input = preprocessor(self.image, return_tensors='pt') example = dict(encoded_input) elif 'xclip' in mi.tags: - from transformers import XCLIPVisionModel - model = XCLIPVisionModel.from_pretrained(name, **model_kwargs) # needs video as input example = {'pixel_values': torch.randn(*(16, 3, 224, 224), dtype=torch.float32)} elif 'audio-spectrogram-transformer' in mi.tags: example = {'input_values': torch.randn(*(1, 1024, 128), dtype=torch.float32)} elif 'mega' in mi.tags: - from transformers import AutoModel - model = AutoModel.from_pretrained(name, **model_kwargs) model.config.output_attentions = True model.config.output_hidden_states = True model.config.return_dict = True example = dict(model.dummy_inputs) elif 'bros' in mi.tags: - from transformers import AutoProcessor, AutoModel - processor = AutoProcessor.from_pretrained(name) - model = AutoModel.from_pretrained(name, **model_kwargs) encoding = processor("to the moon!", return_tensors="pt") bbox = torch.randn([1, 6, 8], dtype=torch.float32) example = dict(input_ids=encoding["input_ids"], bbox=bbox, attention_mask=encoding["attention_mask"]) elif 'upernet' in mi.tags: - from transformers import AutoProcessor, UperNetForSemanticSegmentation - processor = AutoProcessor.from_pretrained(name) - model = UperNetForSemanticSegmentation.from_pretrained(name, **model_kwargs) example = dict(processor(images=self.image, return_tensors="pt")) - elif 'deformable_detr' in mi.tags or 'universal-image-segmentation' in mi.tags: - from transformers import AutoProcessor, AutoModel - + elif 'deformable_detr' in mi.tags or 'oneformer' in mi.tags: processor = AutoProcessor.from_pretrained(name) - model = AutoModel.from_pretrained(name, **model_kwargs) example = dict(processor(images=self.image, task_inputs=["semantic"], return_tensors="pt")) elif 'clap' in mi.tags: - from transformers import AutoModel - model = AutoModel.from_pretrained(name) - - import torch example_inputs_map = { 'audio_model': {'input_features': torch.randn([1, 1, 1001, 64], dtype=torch.float32)}, 'audio_projection': {'hidden_states': torch.randn([1, 768], dtype=torch.float32)}, } - model = model._modules[name_suffix] example = example_inputs_map[name_suffix] elif 'git' in mi.tags: - from transformers import AutoProcessor, AutoModelForCausalLM processor = AutoProcessor.from_pretrained(name) - model = AutoModelForCausalLM.from_pretrained(name) - import torch example = {'pixel_values': torch.randn(*(1, 3, 224, 224), dtype=torch.float32), 'input_ids': torch.randint(1, 100, size=(1, 13), dtype=torch.int64)} elif 'blip-2' in mi.tags: - from transformers import AutoProcessor, AutoModelForVisualQuestionAnswering - processor = AutoProcessor.from_pretrained(name) - model = AutoModelForVisualQuestionAnswering.from_pretrained(name) - example = dict(processor(images=self.image, return_tensors="pt")) - import torch example_inputs_map = { 'vision_model' : {'pixel_values': torch.randn([1, 3, 224, 224], dtype=torch.float32)}, 'qformer': {'query_embeds' : torch.randn([1, 32, 768], dtype=torch.float32), @@ -202,10 +175,8 @@ class TestTransformersModel(TestTorchConvertModel): 'encoder_attention_mask' : torch.ones([1, 257], dtype=torch.int64)}, 'language_projection': {'input' : torch.randn([1, 32, 768], dtype=torch.float32)}, } - model = model._modules[name_suffix] example = example_inputs_map[name_suffix] elif "t5" in mi.tags: - from transformers import T5Tokenizer tokenizer = T5Tokenizer.from_pretrained(name) encoder = tokenizer( "Studies have been shown that owning a dog is good for you", return_tensors="pt") @@ -216,26 +187,16 @@ class TestTransformersModel(TestTorchConvertModel): wav_input_16khz = torch.randn(1, 10000) example = (wav_input_16khz,) elif "vit-gpt2" in name: - from transformers import VisionEncoderDecoderModel, ViTImageProcessor model = VisionEncoderDecoderModel.from_pretrained( name, **model_kwargs) feature_extractor = ViTImageProcessor.from_pretrained(name) encoded_input = feature_extractor( images=[self.image], return_tensors="pt") - class VIT_GPT2_Model(torch.nn.Module): - def __init__(self, model): - super().__init__() - self.model = model - - def forward(self, x): - return self.model.generate(x, max_length=16, num_beams=4) - - model = VIT_GPT2_Model(model) - example = (encoded_input.pixel_values,) + example = dict(encoded_input) + example["decoder_input_ids"] = torch.randint(0, 1000, [1, 20]) + example["decoder_attention_mask"] = torch.ones([1, 20], dtype=torch.int64) elif 'idefics' in mi.tags: - from transformers import IdeficsForVisionText2Text, AutoProcessor - model = IdeficsForVisionText2Text.from_pretrained(name) processor = AutoProcessor.from_pretrained(name) prompts = [[ @@ -253,77 +214,34 @@ class TestTransformersModel(TestTorchConvertModel): ]] inputs = processor(prompts, add_end_of_utterance_token=False, return_tensors="pt") - exit_condition = processor.tokenizer("", add_special_tokens=False).input_ids - bad_words_ids = processor.tokenizer(["", ""], add_special_tokens=False).input_ids - example = dict(inputs) - example.update({ - 'eos_token_id': exit_condition, - 'bad_words_ids': bad_words_ids, - }) - - class Decorator(torch.nn.Module): - def __init__(self, model): - super().__init__() - self.model = model - def forward(self, input_ids, attention_mask, pixel_values, image_attention_mask, eos_token_id, bad_words_ids): - return self.model.generate( - input_ids=input_ids, - attention_mask=attention_mask, - pixel_values=pixel_values, - image_attention_mask=image_attention_mask, - eos_token_id=eos_token_id, - bad_words_ids=bad_words_ids, - max_length=100 - ) - model = Decorator(model) elif 'blip' in mi.tags and 'text2text-generation' in mi.tags: - from transformers import BlipProcessor, BlipForConditionalGeneration - processor = BlipProcessor.from_pretrained(name) - model = BlipForConditionalGeneration.from_pretrained(name) + model = BlipForConditionalGeneration.from_pretrained(name, **model_kwargs) text = "a photography of" inputs = processor(self.image, text, return_tensors="pt") - - class DecoratorForBlipForConditional(torch.nn.Module): - def __init__(self, model): - super().__init__() - self.model = model - - def forward(self, pixel_values, input_ids, attention_mask): - return self.model.generate(pixel_values, input_ids, attention_mask) - - model = DecoratorForBlipForConditional(model) example = dict(inputs) elif 'speecht5' in mi.tags: - from transformers import SpeechT5Processor, SpeechT5ForTextToSpeech, SpeechT5HifiGan from datasets import load_dataset processor = SpeechT5Processor.from_pretrained(name) - model = SpeechT5ForTextToSpeech.from_pretrained(name) + model = SpeechT5ForTextToSpeech.from_pretrained(name, **model_kwargs) inputs = processor(text="Hello, my dog is cute.", return_tensors="pt") # load xvector containing speaker's voice characteristics from a dataset embeddings_dataset = load_dataset("Matthijs/cmu-arctic-xvectors", split="validation") speaker_embeddings = torch.tensor(embeddings_dataset[7306]["xvector"]).unsqueeze(0) - example = {'input_ids': inputs["input_ids"], 'speaker_embeddings': speaker_embeddings} - class DecoratorModelForSeq2SeqLM(torch.nn.Module): - def __init__(self, model): - super().__init__() - self.model = model - def forward(self, input_ids, speaker_embeddings): - return self.model.generate_speech(input_ids=input_ids, speaker_embeddings=speaker_embeddings) #, vocoder=vocoder) - model = DecoratorModelForSeq2SeqLM(model) + example = dict(inputs) + example['speaker_embeddings'] = speaker_embeddings + example['decoder_input_values'] = torch.randn([1, 20, model.config.num_mel_bins]) elif 'layoutlmv2' in mi.tags: - from transformers import LayoutLMv2Processor processor = LayoutLMv2Processor.from_pretrained(name) question = "What's the content of this image?" encoding = processor(self.image, question, max_length=512, truncation=True, return_tensors="pt") example = dict(encoding) elif 'pix2struct' in mi.tags: - from transformers import AutoProcessor, Pix2StructForConditionalGeneration model = Pix2StructForConditionalGeneration.from_pretrained(name, **model_kwargs) processor = AutoProcessor.from_pretrained(name) @@ -334,27 +252,15 @@ class TestTransformersModel(TestTorchConvertModel): question = "What does the label 15 represent? (1) lava (2) core (3) tunnel (4) ash cloud" inputs = processor(images=image, text=question, return_tensors="pt") example = dict(inputs) - - class DecoratorModelForSeq2SeqLM(torch.nn.Module): - def __init__(self, model): - super().__init__() - self.model = model - def forward(self, flattened_patches, attention_mask): - return self.model.generate(flattened_patches=flattened_patches, attention_mask=attention_mask) - model = DecoratorModelForSeq2SeqLM(model) + example["decoder_input_ids"] = torch.randint(0, 1000, [1, 20]) + example["decoder_attention_mask"] = torch.ones([1, 20], dtype=torch.int64) elif "mms-lid" in name: - # mms-lid model config does not have auto_model attribute, only direct loading available - from transformers import Wav2Vec2ForSequenceClassification, AutoFeatureExtractor - model = Wav2Vec2ForSequenceClassification.from_pretrained( - name, **model_kwargs) processor = AutoFeatureExtractor.from_pretrained(name) input_values = processor(torch.randn(16000).numpy(), sampling_rate=16_000, return_tensors="pt") example = {"input_values": input_values.input_values} elif "retribert" in mi.tags: - from transformers import RetriBertTokenizer - text = "How many cats are there?" tokenizer = RetriBertTokenizer.from_pretrained(name) encoding1 = tokenizer( "How many cats are there?", return_tensors="pt") @@ -362,26 +268,22 @@ class TestTransformersModel(TestTorchConvertModel): example = (encoding1.input_ids, encoding1.attention_mask, encoding2.input_ids, encoding2.attention_mask) elif "mgp-str" in mi.tags or "clip_vision_model" in mi.tags: - from transformers import AutoProcessor processor = AutoProcessor.from_pretrained(name) encoded_input = processor(images=self.image, return_tensors="pt") example = (encoded_input.pixel_values,) elif "flava" in mi.tags: - from transformers import AutoProcessor processor = AutoProcessor.from_pretrained(name) encoded_input = processor(text=["a photo of a cat", "a photo of a dog"], images=[self.image, self.image], return_tensors="pt") example = dict(encoded_input) elif "vivit" in mi.tags: - from transformers import VivitImageProcessor frames = list(torch.randint( 0, 255, [32, 3, 224, 224]).to(torch.float32)) processor = VivitImageProcessor.from_pretrained(name) encoded_input = processor(images=frames, return_tensors="pt") example = (encoded_input.pixel_values,) elif "tvlt" in mi.tags: - from transformers import AutoProcessor processor = AutoProcessor.from_pretrained(name) num_frames = 8 images = list(torch.rand(num_frames, 3, 224, 224)) @@ -390,27 +292,23 @@ class TestTransformersModel(TestTorchConvertModel): images, audio, sampling_rate=44100, return_tensors="pt") example = dict(input_dict) elif "gptsan-japanese" in mi.tags: - from transformers import AutoTokenizer processor = AutoTokenizer.from_pretrained(name) text = "織田信長は、" encoded_input = processor(text=[text], return_tensors="pt") example = dict(input_ids=encoded_input.input_ids, token_type_ids=encoded_input.token_type_ids) elif "videomae" in mi.tags or "timesformer" in mi.tags: - from transformers import AutoProcessor processor = AutoProcessor.from_pretrained(name) video = list(torch.randint( 0, 255, [16, 3, 224, 224]).to(torch.float32)) inputs = processor(video, return_tensors="pt") example = dict(inputs) elif 'text-to-speech' in mi.tags: - from transformers import AutoTokenizer tokenizer = AutoTokenizer.from_pretrained(name) text = "some example text in the English language" inputs = tokenizer(text, return_tensors="pt") example = dict(inputs) elif 'musicgen' in mi.tags: - from transformers import AutoProcessor, AutoModelForTextToWaveform processor = AutoProcessor.from_pretrained(name) model = AutoModelForTextToWaveform.from_pretrained(name, **model_kwargs) @@ -425,7 +323,6 @@ class TestTransformersModel(TestTorchConvertModel): example["decoder_input_ids"] = torch.ones( (inputs.input_ids.shape[0] * model.decoder.num_codebooks, 1), dtype=torch.long) * pad_token_id elif 'kosmos-2' in mi.tags: - from transformers import AutoProcessor processor = AutoProcessor.from_pretrained(name) prompt = "An image of" @@ -434,37 +331,24 @@ class TestTransformersModel(TestTorchConvertModel): else: try: if auto_model == "AutoModelForCausalLM": - from transformers import AutoConfig, AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained(name) - model = AutoModelForCausalLM.from_pretrained( - name, **model_kwargs) text = "Replace me by any text you'd like." encoded_input = tokenizer(text, return_tensors='pt') - inputs_dict = dict(encoded_input) - if "facebook/incoder" in name and "token_type_ids" in inputs_dict: - del inputs_dict["token_type_ids"] - example = inputs_dict + example = dict(encoded_input) + if "facebook/incoder" in name and "token_type_ids" in example: + del example["token_type_ids"] elif auto_model == "AutoModelForMaskedLM": - from transformers import AutoTokenizer, AutoModelForMaskedLM tokenizer = AutoTokenizer.from_pretrained(name) - model = AutoModelForMaskedLM.from_pretrained( - name, **model_kwargs) text = "Replace me by any text you'd like." encoded_input = tokenizer(text, return_tensors='pt') example = dict(encoded_input) elif auto_model == "AutoModelForImageClassification": - from transformers import AutoProcessor, AutoModelForImageClassification processor = AutoProcessor.from_pretrained(name) - model = AutoModelForImageClassification.from_pretrained( - name, **model_kwargs) encoded_input = processor( images=self.image, return_tensors="pt") example = dict(encoded_input) elif auto_model == "AutoModelForSeq2SeqLM": - from transformers import AutoTokenizer, AutoModelForSeq2SeqLM tokenizer = AutoTokenizer.from_pretrained(name) - model = AutoModelForSeq2SeqLM.from_pretrained( - name, **model_kwargs) inputs = tokenizer( "Studies have been shown that owning a dog is good for you", return_tensors="pt") decoder_inputs = tokenizer( @@ -475,30 +359,21 @@ class TestTransformersModel(TestTorchConvertModel): example = dict(input_ids=inputs.input_ids, decoder_input_ids=decoder_inputs.input_ids) elif auto_model == "AutoModelForSpeechSeq2Seq": - from transformers import AutoProcessor, AutoModelForSpeechSeq2Seq from datasets import load_dataset processor = AutoProcessor.from_pretrained(name) - model = AutoModelForSpeechSeq2Seq.from_pretrained( - name, **model_kwargs) inputs = processor(torch.randn(1000).numpy(), sampling_rate=16000, return_tensors="pt") example = dict(inputs) elif auto_model == "AutoModelForCTC": - from transformers import AutoProcessor, AutoModelForCTC from datasets import load_dataset processor = AutoProcessor.from_pretrained(name) - model = AutoModelForCTC.from_pretrained( - name, **model_kwargs) input_values = processor(torch.randn(1000).numpy(), return_tensors="pt") example = dict(input_values) elif auto_model == "AutoModelForTableQuestionAnswering": import pandas as pd - from transformers import AutoTokenizer, AutoModelForTableQuestionAnswering tokenizer = AutoTokenizer.from_pretrained(name) - model = AutoModelForTableQuestionAnswering.from_pretrained( - name, **model_kwargs) data = {"Actors": ["Brad Pitt", "Leonardo Di Caprio", "George Clooney"], "Number of movies": ["87", "53", "69"]} queries = ["What is the name of the first actor?", @@ -514,7 +389,6 @@ class TestTransformersModel(TestTorchConvertModel): token_type_ids=encoded_input["token_type_ids"], attention_mask=encoded_input["attention_mask"]) else: - from transformers import AutoTokenizer, AutoProcessor text = "Replace me by any text you'd like." if auto_processor is not None and "Tokenizer" not in auto_processor: processor = AutoProcessor.from_pretrained(name) @@ -527,10 +401,11 @@ class TestTransformersModel(TestTorchConvertModel): except: pass if model is None: - from transformers import AutoModel - model = AutoModel.from_pretrained(name, **model_kwargs) + model = self.load_model_with_default_class(name, **model_kwargs) if hasattr(model, "set_default_language"): model.set_default_language("en_XX") + if name_suffix != '': + model = model._modules[name_suffix] if example is None: if "encodec" in mi.tags: example = (torch.randn(1, 1, 100),) @@ -557,13 +432,25 @@ class TestTransformersModel(TestTorchConvertModel): self.cuda_available, self.gptq_postinit = None, None super().teardown_method() + @staticmethod + def load_model_with_default_class(name, **kwargs): + try: + mi = model_info(name) + assert len({"owlv2", "owlvit", "vit_mae"}.intersection(mi.tags)) == 0, "TBD: support default classes of these models" + assert "architectures" in mi.config and len(mi.config["architectures"]) == 1 + class_name = mi.config["architectures"][0] + model_class = transformers.__getattr__(class_name) + return model_class.from_pretrained(name, **kwargs) + except: + return AutoModel.from_pretrained(name, **kwargs) + @pytest.mark.parametrize("name,type", [("allenai/led-base-16384", "led"), ("bert-base-uncased", "bert"), ("google/flan-t5-base", "t5"), ("google/tapas-large-finetuned-wtq", "tapas"), ("gpt2", "gpt2"), ("openai/clip-vit-large-patch14", "clip"), - ("OpenVINO/opt-125m-gptq", "opt") + ("OpenVINO/opt-125m-gptq", "opt"), ]) @pytest.mark.precommit def test_convert_model_precommit(self, name, type, ie_device):