Refactor hf GHA tests (#24048)
### Details: - *Refactor hf tests* ### Tickets: - *ticket-id*
This commit is contained in:
parent
eb025fad1e
commit
03fa47a2ff
|
|
@ -23,7 +23,7 @@ asapp/sew-d-base-plus-400k-ft-ls100h,sew-d
|
|||
ashishpatel26/span-marker-bert-base-fewnerd-coarse-super,span-marker,skip,Load problem
|
||||
asi/albert-act-tiny,albert_act,skip,Load problem
|
||||
BAAI/AltCLIP,altclip
|
||||
BAAI/AquilaCode-py,aquila,skip,Load problem
|
||||
BAAI/AquilaCode-py,aquila
|
||||
bana513/opennmt-translator-en-hu,opennmt-translator,skip,Load problem
|
||||
benjamin/wtp-bert-mini,bert-char,skip,Load problem
|
||||
benjamin/wtp-canine-s-1l,la-canine,skip,Load problem
|
||||
|
|
@ -79,6 +79,7 @@ facebook/m2m100_418M,m2m_100
|
|||
facebook/mask2former-swin-base-coco-panoptic,mask2former
|
||||
facebook/maskformer-swin-base-coco,maskformer
|
||||
facebook/mbart-large-50-many-to-many-mmt,mbart
|
||||
facebook/mms-lid-126,wav2vec2
|
||||
facebook/mms-tts-eng,vits,xfail,Accuracy failed: results cannot be broadcasted
|
||||
facebook/musicgen-small,musicgen
|
||||
facebook/opt-125m,opt
|
||||
|
|
@ -90,7 +91,7 @@ facebook/wmt19-ru-en,fsmt,xfail,Tracing problem
|
|||
facebook/xglm-7.5B,xglm
|
||||
facebook/xlm-roberta-xl,xlm-roberta-xl
|
||||
facebook/xmod-base,xmod
|
||||
flax-community/ft5-cnn-dm,f_t5,skip,Load problem
|
||||
flax-community/ft5-cnn-dm,f_t5
|
||||
fnlp/elasticbert-base,elasticbert,skip,Load problem
|
||||
FranzStrauss/ponet-base-uncased,ponet,skip,Load problem
|
||||
funnel-transformer/small,funnel
|
||||
|
|
@ -117,7 +118,7 @@ google/reformer-crime-and-punishment,reformer,xfail,Tracing problem
|
|||
google/tapas-large-finetuned-wtq,tapas
|
||||
google/vit-hybrid-base-bit-384,vit-hybrid,skip,Load problem
|
||||
google/vivit-b-16x2-kinetics400,vivit
|
||||
Goutham-Vignesh/ContributionSentClassification-scibert,scibert,skip,Load problem
|
||||
Goutham-Vignesh/ContributionSentClassification-scibert,scibert
|
||||
gpt2,gpt2
|
||||
Graphcore/groupbert-base-uncased,groupbert,skip,Load problem
|
||||
haoranzhao419/saffu-100M-0.1,saffu-100M-0.1,skip,Load problem
|
||||
|
|
@ -166,7 +167,7 @@ HJHGJGHHG/GAU-Base-Full,gau,skip,Load problem
|
|||
huggingface/autoformer-tourism-monthly,autoformer,skip,Load problem
|
||||
huggingface/informer-tourism-monthly,informer,skip,Load problem
|
||||
huggingface/time-series-transformer-tourism-monthly,time_series_transformer,skip,Load problem
|
||||
HuggingFaceM4/tiny-random-idefics,idefics,xfail,tracing error: Please check correctness of provided example_input (eval was correct but trace failed with incommatible tuples and tensors)
|
||||
HuggingFaceM4/tiny-random-idefics,idefics,xfail,Unsupported op aten::any aten::einsum prim::TupleConstruct prim::TupleUnpack
|
||||
HuggingFaceM4/tiny-random-vllama-clip,vllama,skip,Load problem
|
||||
HuggingFaceM4/tiny-random-vopt-clip,vopt,skip,Load problem
|
||||
HuiHuang/gpt3-damo-base-zh,gpt3,skip,Load problem
|
||||
|
|
@ -184,7 +185,6 @@ jambran/depression-classification,DepressionDetection,skip,Load problem
|
|||
Jellywibble/dalio-reward-charlie-v1,reward-model,skip,Load problem
|
||||
JonasGeiping/crammed-bert-legacy,crammedBERT,skip,Load problem
|
||||
jonatasgrosman/wav2vec2-large-xlsr-53-english,wav2vec2,xfail,Unsupported op aten::index_put_ prim::TupleConstruct
|
||||
facebook/mms-lid-126,wav2vec2
|
||||
Joqsan/test-my-fnet,my_fnet,skip,Load problem
|
||||
jozhang97/deta-swin-large,deta,skip,Load problem
|
||||
jploski/retnet-mini-shakespeare,retnet,skip,Load problem
|
||||
|
|
@ -247,7 +247,7 @@ microsoft/markuplm-base,markuplm
|
|||
microsoft/prophetnet-large-uncased-squad-qg,prophetnet
|
||||
microsoft/resnet-50,resnet
|
||||
microsoft/speecht5_hifigan,hifigan,skip,Load problem
|
||||
microsoft/speecht5_tts,speecht5,xfail,Tracing error: hangs with no error (probably because of infinite while inside generate)
|
||||
microsoft/speecht5_tts,speecht5,xfail,Unsupported op aten::bernoulli
|
||||
microsoft/swinv2-tiny-patch4-window8-256,swinv2
|
||||
microsoft/table-transformer-detection,table-transformer
|
||||
microsoft/unispeech-1350-en-17h-ky-ft-1h,unispeech
|
||||
|
|
@ -303,7 +303,7 @@ paulhindemith/test-zeroshot,test-zeroshot,skip,Load problem
|
|||
PGT/orig-nystromformer-s-artificial-balanced-max500-490000-0,graph_nystromformer,skip,Load problem
|
||||
pie/example-ner-spanclf-conll03,TransformerSpanClassificationModel,skip,Load problem
|
||||
pie/example-re-textclf-tacred,TransformerTextClassificationModel,skip,Load problem
|
||||
pleisto/yuren-baichuan-7b,multimodal_llama,skip,Load problem
|
||||
pleisto/yuren-baichuan-7b,multimodal_llama
|
||||
predictia/europe_reanalysis_downscaler_convbaseline,convbilinear,skip,Load problem
|
||||
predictia/europe_reanalysis_downscaler_convswin2sr,conv_swin2sr,skip,Load problem
|
||||
pszemraj/led-large-book-summary,led
|
||||
|
|
@ -341,7 +341,7 @@ shikhartuli/flexibert-mini,flexibert,skip,Load problem
|
|||
shikras/shikra-7b-delta-v1-0708,shikra,skip,Load problem
|
||||
shi-labs/dinat-mini-in1k-224,dinat,xfail,Accuracy validation failed
|
||||
shi-labs/nat-mini-in1k-224,nat,xfail,Accuracy validation failed
|
||||
shi-labs/oneformer_ade20k_swin_large,oneformer,xfail,Tracing error: Please check correctness of provided example_input (but eval was correct)
|
||||
shi-labs/oneformer_ade20k_swin_large,oneformer,xfail,Different number of outputs between framework and OpenVINO
|
||||
shuqi/seed-encoder,seed_encoder,skip,Load problem
|
||||
sijunhe/nezha-cn-base,nezha
|
||||
sjiang1/codecse,roberta_for_cl,skip,Load problem
|
||||
|
|
@ -352,7 +352,7 @@ solotimes/lavibe_base,donut,skip,Load problem
|
|||
songlab/gpn-brassicales,ConvNet,skip,Load problem
|
||||
speechbrain/m-ctc-t-large,mctct
|
||||
Splend1dchan/wav2vec2-large-lv60_t5lephone-small_lna_bs64,speechmix,skip,Load problem
|
||||
stefan-it/bort-full,bort,skip,Load problem
|
||||
stefan-it/bort-full,bort
|
||||
SteveZhan/my-resnet50d,resnet_steve,skip,Load problem
|
||||
suno/bark,bark,skip,Load problem
|
||||
surajnair/r3m-50,r3m,skip,Load problem
|
||||
|
|
|
|||
|
|
@ -8,9 +8,12 @@ import torch
|
|||
from huggingface_hub import model_info
|
||||
from models_hub_common.constants import hf_hub_cache_dir
|
||||
from models_hub_common.utils import cleanup_dir
|
||||
import transformers
|
||||
from transformers import AutoConfig, AutoModel, AutoProcessor, AutoTokenizer, AutoFeatureExtractor, AutoModelForTextToWaveform, \
|
||||
CLIPFeatureExtractor, XCLIPVisionModel, T5Tokenizer, VisionEncoderDecoderModel, ViTImageProcessor, BlipProcessor, BlipForConditionalGeneration, \
|
||||
SpeechT5Processor, SpeechT5ForTextToSpeech, LayoutLMv2Processor, Pix2StructForConditionalGeneration, RetriBertTokenizer, VivitImageProcessor
|
||||
|
||||
from torch_utils import TestTorchConvertModel
|
||||
from torch_utils import process_pytest_marks
|
||||
from torch_utils import TestTorchConvertModel, process_pytest_marks
|
||||
|
||||
def is_gptq_model(config):
|
||||
config_dict = config.to_dict() if not isinstance(config, dict) else config
|
||||
|
|
@ -92,16 +95,14 @@ class TestTransformersModel(TestTorchConvertModel):
|
|||
from PIL import Image
|
||||
import requests
|
||||
|
||||
self.infer_timeout = 1000
|
||||
self.infer_timeout = 1800
|
||||
|
||||
url = "http://images.cocodataset.org/val2017/000000039769.jpg"
|
||||
self.image = Image.open(requests.get(url, stream=True).raw)
|
||||
self.cuda_available, self.gptq_postinit = None, None
|
||||
|
||||
def load_model(self, name, type):
|
||||
import torch
|
||||
name_suffix = ''
|
||||
from transformers import AutoConfig
|
||||
if name.find(':') != -1:
|
||||
name_suffix = name[name.find(':') + 1:]
|
||||
name = name[:name.find(':')]
|
||||
|
|
@ -128,73 +129,45 @@ class TestTransformersModel(TestTorchConvertModel):
|
|||
except:
|
||||
auto_model = None
|
||||
if "clip_vision_model" in mi.tags:
|
||||
from transformers import CLIPVisionModel, CLIPFeatureExtractor
|
||||
model = CLIPVisionModel.from_pretrained(name, torchscript=True)
|
||||
preprocessor = CLIPFeatureExtractor.from_pretrained(name)
|
||||
encoded_input = preprocessor(self.image, return_tensors='pt')
|
||||
example = dict(encoded_input)
|
||||
elif 'xclip' in mi.tags:
|
||||
from transformers import XCLIPVisionModel
|
||||
|
||||
model = XCLIPVisionModel.from_pretrained(name, **model_kwargs)
|
||||
# needs video as input
|
||||
example = {'pixel_values': torch.randn(*(16, 3, 224, 224), dtype=torch.float32)}
|
||||
elif 'audio-spectrogram-transformer' in mi.tags:
|
||||
example = {'input_values': torch.randn(*(1, 1024, 128), dtype=torch.float32)}
|
||||
elif 'mega' in mi.tags:
|
||||
from transformers import AutoModel
|
||||
|
||||
model = AutoModel.from_pretrained(name, **model_kwargs)
|
||||
model.config.output_attentions = True
|
||||
model.config.output_hidden_states = True
|
||||
model.config.return_dict = True
|
||||
example = dict(model.dummy_inputs)
|
||||
elif 'bros' in mi.tags:
|
||||
from transformers import AutoProcessor, AutoModel
|
||||
|
||||
processor = AutoProcessor.from_pretrained(name)
|
||||
model = AutoModel.from_pretrained(name, **model_kwargs)
|
||||
encoding = processor("to the moon!", return_tensors="pt")
|
||||
bbox = torch.randn([1, 6, 8], dtype=torch.float32)
|
||||
example = dict(input_ids=encoding["input_ids"], bbox=bbox, attention_mask=encoding["attention_mask"])
|
||||
elif 'upernet' in mi.tags:
|
||||
from transformers import AutoProcessor, UperNetForSemanticSegmentation
|
||||
|
||||
processor = AutoProcessor.from_pretrained(name)
|
||||
model = UperNetForSemanticSegmentation.from_pretrained(name, **model_kwargs)
|
||||
example = dict(processor(images=self.image, return_tensors="pt"))
|
||||
elif 'deformable_detr' in mi.tags or 'universal-image-segmentation' in mi.tags:
|
||||
from transformers import AutoProcessor, AutoModel
|
||||
|
||||
elif 'deformable_detr' in mi.tags or 'oneformer' in mi.tags:
|
||||
processor = AutoProcessor.from_pretrained(name)
|
||||
model = AutoModel.from_pretrained(name, **model_kwargs)
|
||||
example = dict(processor(images=self.image, task_inputs=["semantic"], return_tensors="pt"))
|
||||
elif 'clap' in mi.tags:
|
||||
from transformers import AutoModel
|
||||
model = AutoModel.from_pretrained(name)
|
||||
|
||||
import torch
|
||||
example_inputs_map = {
|
||||
'audio_model': {'input_features': torch.randn([1, 1, 1001, 64], dtype=torch.float32)},
|
||||
'audio_projection': {'hidden_states': torch.randn([1, 768], dtype=torch.float32)},
|
||||
}
|
||||
model = model._modules[name_suffix]
|
||||
example = example_inputs_map[name_suffix]
|
||||
elif 'git' in mi.tags:
|
||||
from transformers import AutoProcessor, AutoModelForCausalLM
|
||||
processor = AutoProcessor.from_pretrained(name)
|
||||
model = AutoModelForCausalLM.from_pretrained(name)
|
||||
import torch
|
||||
example = {'pixel_values': torch.randn(*(1, 3, 224, 224), dtype=torch.float32),
|
||||
'input_ids': torch.randint(1, 100, size=(1, 13), dtype=torch.int64)}
|
||||
elif 'blip-2' in mi.tags:
|
||||
from transformers import AutoProcessor, AutoModelForVisualQuestionAnswering
|
||||
|
||||
processor = AutoProcessor.from_pretrained(name)
|
||||
model = AutoModelForVisualQuestionAnswering.from_pretrained(name)
|
||||
|
||||
example = dict(processor(images=self.image, return_tensors="pt"))
|
||||
import torch
|
||||
example_inputs_map = {
|
||||
'vision_model' : {'pixel_values': torch.randn([1, 3, 224, 224], dtype=torch.float32)},
|
||||
'qformer': {'query_embeds' : torch.randn([1, 32, 768], dtype=torch.float32),
|
||||
|
|
@ -202,10 +175,8 @@ class TestTransformersModel(TestTorchConvertModel):
|
|||
'encoder_attention_mask' : torch.ones([1, 257], dtype=torch.int64)},
|
||||
'language_projection': {'input' : torch.randn([1, 32, 768], dtype=torch.float32)},
|
||||
}
|
||||
model = model._modules[name_suffix]
|
||||
example = example_inputs_map[name_suffix]
|
||||
elif "t5" in mi.tags:
|
||||
from transformers import T5Tokenizer
|
||||
tokenizer = T5Tokenizer.from_pretrained(name)
|
||||
encoder = tokenizer(
|
||||
"Studies have been shown that owning a dog is good for you", return_tensors="pt")
|
||||
|
|
@ -216,26 +187,16 @@ class TestTransformersModel(TestTorchConvertModel):
|
|||
wav_input_16khz = torch.randn(1, 10000)
|
||||
example = (wav_input_16khz,)
|
||||
elif "vit-gpt2" in name:
|
||||
from transformers import VisionEncoderDecoderModel, ViTImageProcessor
|
||||
model = VisionEncoderDecoderModel.from_pretrained(
|
||||
name, **model_kwargs)
|
||||
feature_extractor = ViTImageProcessor.from_pretrained(name)
|
||||
encoded_input = feature_extractor(
|
||||
images=[self.image], return_tensors="pt")
|
||||
|
||||
class VIT_GPT2_Model(torch.nn.Module):
|
||||
def __init__(self, model):
|
||||
super().__init__()
|
||||
self.model = model
|
||||
|
||||
def forward(self, x):
|
||||
return self.model.generate(x, max_length=16, num_beams=4)
|
||||
|
||||
model = VIT_GPT2_Model(model)
|
||||
example = (encoded_input.pixel_values,)
|
||||
example = dict(encoded_input)
|
||||
example["decoder_input_ids"] = torch.randint(0, 1000, [1, 20])
|
||||
example["decoder_attention_mask"] = torch.ones([1, 20], dtype=torch.int64)
|
||||
elif 'idefics' in mi.tags:
|
||||
from transformers import IdeficsForVisionText2Text, AutoProcessor
|
||||
model = IdeficsForVisionText2Text.from_pretrained(name)
|
||||
processor = AutoProcessor.from_pretrained(name)
|
||||
|
||||
prompts = [[
|
||||
|
|
@ -253,77 +214,34 @@ class TestTransformersModel(TestTorchConvertModel):
|
|||
]]
|
||||
|
||||
inputs = processor(prompts, add_end_of_utterance_token=False, return_tensors="pt")
|
||||
exit_condition = processor.tokenizer("<end_of_utterance>", add_special_tokens=False).input_ids
|
||||
bad_words_ids = processor.tokenizer(["<image>", "<fake_token_around_image>"], add_special_tokens=False).input_ids
|
||||
|
||||
example = dict(inputs)
|
||||
example.update({
|
||||
'eos_token_id': exit_condition,
|
||||
'bad_words_ids': bad_words_ids,
|
||||
})
|
||||
|
||||
class Decorator(torch.nn.Module):
|
||||
def __init__(self, model):
|
||||
super().__init__()
|
||||
self.model = model
|
||||
def forward(self, input_ids, attention_mask, pixel_values, image_attention_mask, eos_token_id, bad_words_ids):
|
||||
return self.model.generate(
|
||||
input_ids=input_ids,
|
||||
attention_mask=attention_mask,
|
||||
pixel_values=pixel_values,
|
||||
image_attention_mask=image_attention_mask,
|
||||
eos_token_id=eos_token_id,
|
||||
bad_words_ids=bad_words_ids,
|
||||
max_length=100
|
||||
)
|
||||
model = Decorator(model)
|
||||
elif 'blip' in mi.tags and 'text2text-generation' in mi.tags:
|
||||
from transformers import BlipProcessor, BlipForConditionalGeneration
|
||||
|
||||
processor = BlipProcessor.from_pretrained(name)
|
||||
model = BlipForConditionalGeneration.from_pretrained(name)
|
||||
model = BlipForConditionalGeneration.from_pretrained(name, **model_kwargs)
|
||||
text = "a photography of"
|
||||
inputs = processor(self.image, text, return_tensors="pt")
|
||||
|
||||
class DecoratorForBlipForConditional(torch.nn.Module):
|
||||
def __init__(self, model):
|
||||
super().__init__()
|
||||
self.model = model
|
||||
|
||||
def forward(self, pixel_values, input_ids, attention_mask):
|
||||
return self.model.generate(pixel_values, input_ids, attention_mask)
|
||||
|
||||
model = DecoratorForBlipForConditional(model)
|
||||
example = dict(inputs)
|
||||
elif 'speecht5' in mi.tags:
|
||||
from transformers import SpeechT5Processor, SpeechT5ForTextToSpeech, SpeechT5HifiGan
|
||||
from datasets import load_dataset
|
||||
|
||||
processor = SpeechT5Processor.from_pretrained(name)
|
||||
model = SpeechT5ForTextToSpeech.from_pretrained(name)
|
||||
model = SpeechT5ForTextToSpeech.from_pretrained(name, **model_kwargs)
|
||||
|
||||
inputs = processor(text="Hello, my dog is cute.", return_tensors="pt")
|
||||
# load xvector containing speaker's voice characteristics from a dataset
|
||||
embeddings_dataset = load_dataset("Matthijs/cmu-arctic-xvectors", split="validation")
|
||||
speaker_embeddings = torch.tensor(embeddings_dataset[7306]["xvector"]).unsqueeze(0)
|
||||
|
||||
example = {'input_ids': inputs["input_ids"], 'speaker_embeddings': speaker_embeddings}
|
||||
class DecoratorModelForSeq2SeqLM(torch.nn.Module):
|
||||
def __init__(self, model):
|
||||
super().__init__()
|
||||
self.model = model
|
||||
def forward(self, input_ids, speaker_embeddings):
|
||||
return self.model.generate_speech(input_ids=input_ids, speaker_embeddings=speaker_embeddings) #, vocoder=vocoder)
|
||||
model = DecoratorModelForSeq2SeqLM(model)
|
||||
example = dict(inputs)
|
||||
example['speaker_embeddings'] = speaker_embeddings
|
||||
example['decoder_input_values'] = torch.randn([1, 20, model.config.num_mel_bins])
|
||||
elif 'layoutlmv2' in mi.tags:
|
||||
from transformers import LayoutLMv2Processor
|
||||
processor = LayoutLMv2Processor.from_pretrained(name)
|
||||
|
||||
question = "What's the content of this image?"
|
||||
encoding = processor(self.image, question, max_length=512, truncation=True, return_tensors="pt")
|
||||
example = dict(encoding)
|
||||
elif 'pix2struct' in mi.tags:
|
||||
from transformers import AutoProcessor, Pix2StructForConditionalGeneration
|
||||
model = Pix2StructForConditionalGeneration.from_pretrained(name, **model_kwargs)
|
||||
processor = AutoProcessor.from_pretrained(name)
|
||||
|
||||
|
|
@ -334,27 +252,15 @@ class TestTransformersModel(TestTorchConvertModel):
|
|||
question = "What does the label 15 represent? (1) lava (2) core (3) tunnel (4) ash cloud"
|
||||
inputs = processor(images=image, text=question, return_tensors="pt")
|
||||
example = dict(inputs)
|
||||
|
||||
class DecoratorModelForSeq2SeqLM(torch.nn.Module):
|
||||
def __init__(self, model):
|
||||
super().__init__()
|
||||
self.model = model
|
||||
def forward(self, flattened_patches, attention_mask):
|
||||
return self.model.generate(flattened_patches=flattened_patches, attention_mask=attention_mask)
|
||||
model = DecoratorModelForSeq2SeqLM(model)
|
||||
example["decoder_input_ids"] = torch.randint(0, 1000, [1, 20])
|
||||
example["decoder_attention_mask"] = torch.ones([1, 20], dtype=torch.int64)
|
||||
elif "mms-lid" in name:
|
||||
# mms-lid model config does not have auto_model attribute, only direct loading available
|
||||
from transformers import Wav2Vec2ForSequenceClassification, AutoFeatureExtractor
|
||||
model = Wav2Vec2ForSequenceClassification.from_pretrained(
|
||||
name, **model_kwargs)
|
||||
processor = AutoFeatureExtractor.from_pretrained(name)
|
||||
input_values = processor(torch.randn(16000).numpy(),
|
||||
sampling_rate=16_000,
|
||||
return_tensors="pt")
|
||||
example = {"input_values": input_values.input_values}
|
||||
elif "retribert" in mi.tags:
|
||||
from transformers import RetriBertTokenizer
|
||||
text = "How many cats are there?"
|
||||
tokenizer = RetriBertTokenizer.from_pretrained(name)
|
||||
encoding1 = tokenizer(
|
||||
"How many cats are there?", return_tensors="pt")
|
||||
|
|
@ -362,26 +268,22 @@ class TestTransformersModel(TestTorchConvertModel):
|
|||
example = (encoding1.input_ids, encoding1.attention_mask,
|
||||
encoding2.input_ids, encoding2.attention_mask)
|
||||
elif "mgp-str" in mi.tags or "clip_vision_model" in mi.tags:
|
||||
from transformers import AutoProcessor
|
||||
processor = AutoProcessor.from_pretrained(name)
|
||||
encoded_input = processor(images=self.image, return_tensors="pt")
|
||||
example = (encoded_input.pixel_values,)
|
||||
elif "flava" in mi.tags:
|
||||
from transformers import AutoProcessor
|
||||
processor = AutoProcessor.from_pretrained(name)
|
||||
encoded_input = processor(text=["a photo of a cat", "a photo of a dog"],
|
||||
images=[self.image, self.image],
|
||||
return_tensors="pt")
|
||||
example = dict(encoded_input)
|
||||
elif "vivit" in mi.tags:
|
||||
from transformers import VivitImageProcessor
|
||||
frames = list(torch.randint(
|
||||
0, 255, [32, 3, 224, 224]).to(torch.float32))
|
||||
processor = VivitImageProcessor.from_pretrained(name)
|
||||
encoded_input = processor(images=frames, return_tensors="pt")
|
||||
example = (encoded_input.pixel_values,)
|
||||
elif "tvlt" in mi.tags:
|
||||
from transformers import AutoProcessor
|
||||
processor = AutoProcessor.from_pretrained(name)
|
||||
num_frames = 8
|
||||
images = list(torch.rand(num_frames, 3, 224, 224))
|
||||
|
|
@ -390,27 +292,23 @@ class TestTransformersModel(TestTorchConvertModel):
|
|||
images, audio, sampling_rate=44100, return_tensors="pt")
|
||||
example = dict(input_dict)
|
||||
elif "gptsan-japanese" in mi.tags:
|
||||
from transformers import AutoTokenizer
|
||||
processor = AutoTokenizer.from_pretrained(name)
|
||||
text = "織田信長は、"
|
||||
encoded_input = processor(text=[text], return_tensors="pt")
|
||||
example = dict(input_ids=encoded_input.input_ids,
|
||||
token_type_ids=encoded_input.token_type_ids)
|
||||
elif "videomae" in mi.tags or "timesformer" in mi.tags:
|
||||
from transformers import AutoProcessor
|
||||
processor = AutoProcessor.from_pretrained(name)
|
||||
video = list(torch.randint(
|
||||
0, 255, [16, 3, 224, 224]).to(torch.float32))
|
||||
inputs = processor(video, return_tensors="pt")
|
||||
example = dict(inputs)
|
||||
elif 'text-to-speech' in mi.tags:
|
||||
from transformers import AutoTokenizer
|
||||
tokenizer = AutoTokenizer.from_pretrained(name)
|
||||
text = "some example text in the English language"
|
||||
inputs = tokenizer(text, return_tensors="pt")
|
||||
example = dict(inputs)
|
||||
elif 'musicgen' in mi.tags:
|
||||
from transformers import AutoProcessor, AutoModelForTextToWaveform
|
||||
processor = AutoProcessor.from_pretrained(name)
|
||||
model = AutoModelForTextToWaveform.from_pretrained(name, **model_kwargs)
|
||||
|
||||
|
|
@ -425,7 +323,6 @@ class TestTransformersModel(TestTorchConvertModel):
|
|||
example["decoder_input_ids"] = torch.ones(
|
||||
(inputs.input_ids.shape[0] * model.decoder.num_codebooks, 1), dtype=torch.long) * pad_token_id
|
||||
elif 'kosmos-2' in mi.tags:
|
||||
from transformers import AutoProcessor
|
||||
processor = AutoProcessor.from_pretrained(name)
|
||||
|
||||
prompt = "<grounding>An image of"
|
||||
|
|
@ -434,37 +331,24 @@ class TestTransformersModel(TestTorchConvertModel):
|
|||
else:
|
||||
try:
|
||||
if auto_model == "AutoModelForCausalLM":
|
||||
from transformers import AutoConfig, AutoTokenizer, AutoModelForCausalLM
|
||||
tokenizer = AutoTokenizer.from_pretrained(name)
|
||||
model = AutoModelForCausalLM.from_pretrained(
|
||||
name, **model_kwargs)
|
||||
text = "Replace me by any text you'd like."
|
||||
encoded_input = tokenizer(text, return_tensors='pt')
|
||||
inputs_dict = dict(encoded_input)
|
||||
if "facebook/incoder" in name and "token_type_ids" in inputs_dict:
|
||||
del inputs_dict["token_type_ids"]
|
||||
example = inputs_dict
|
||||
example = dict(encoded_input)
|
||||
if "facebook/incoder" in name and "token_type_ids" in example:
|
||||
del example["token_type_ids"]
|
||||
elif auto_model == "AutoModelForMaskedLM":
|
||||
from transformers import AutoTokenizer, AutoModelForMaskedLM
|
||||
tokenizer = AutoTokenizer.from_pretrained(name)
|
||||
model = AutoModelForMaskedLM.from_pretrained(
|
||||
name, **model_kwargs)
|
||||
text = "Replace me by any text you'd like."
|
||||
encoded_input = tokenizer(text, return_tensors='pt')
|
||||
example = dict(encoded_input)
|
||||
elif auto_model == "AutoModelForImageClassification":
|
||||
from transformers import AutoProcessor, AutoModelForImageClassification
|
||||
processor = AutoProcessor.from_pretrained(name)
|
||||
model = AutoModelForImageClassification.from_pretrained(
|
||||
name, **model_kwargs)
|
||||
encoded_input = processor(
|
||||
images=self.image, return_tensors="pt")
|
||||
example = dict(encoded_input)
|
||||
elif auto_model == "AutoModelForSeq2SeqLM":
|
||||
from transformers import AutoTokenizer, AutoModelForSeq2SeqLM
|
||||
tokenizer = AutoTokenizer.from_pretrained(name)
|
||||
model = AutoModelForSeq2SeqLM.from_pretrained(
|
||||
name, **model_kwargs)
|
||||
inputs = tokenizer(
|
||||
"Studies have been shown that owning a dog is good for you", return_tensors="pt")
|
||||
decoder_inputs = tokenizer(
|
||||
|
|
@ -475,30 +359,21 @@ class TestTransformersModel(TestTorchConvertModel):
|
|||
example = dict(input_ids=inputs.input_ids,
|
||||
decoder_input_ids=decoder_inputs.input_ids)
|
||||
elif auto_model == "AutoModelForSpeechSeq2Seq":
|
||||
from transformers import AutoProcessor, AutoModelForSpeechSeq2Seq
|
||||
from datasets import load_dataset
|
||||
processor = AutoProcessor.from_pretrained(name)
|
||||
model = AutoModelForSpeechSeq2Seq.from_pretrained(
|
||||
name, **model_kwargs)
|
||||
inputs = processor(torch.randn(1000).numpy(),
|
||||
sampling_rate=16000,
|
||||
return_tensors="pt")
|
||||
example = dict(inputs)
|
||||
elif auto_model == "AutoModelForCTC":
|
||||
from transformers import AutoProcessor, AutoModelForCTC
|
||||
from datasets import load_dataset
|
||||
processor = AutoProcessor.from_pretrained(name)
|
||||
model = AutoModelForCTC.from_pretrained(
|
||||
name, **model_kwargs)
|
||||
input_values = processor(torch.randn(1000).numpy(),
|
||||
return_tensors="pt")
|
||||
example = dict(input_values)
|
||||
elif auto_model == "AutoModelForTableQuestionAnswering":
|
||||
import pandas as pd
|
||||
from transformers import AutoTokenizer, AutoModelForTableQuestionAnswering
|
||||
tokenizer = AutoTokenizer.from_pretrained(name)
|
||||
model = AutoModelForTableQuestionAnswering.from_pretrained(
|
||||
name, **model_kwargs)
|
||||
data = {"Actors": ["Brad Pitt", "Leonardo Di Caprio", "George Clooney"],
|
||||
"Number of movies": ["87", "53", "69"]}
|
||||
queries = ["What is the name of the first actor?",
|
||||
|
|
@ -514,7 +389,6 @@ class TestTransformersModel(TestTorchConvertModel):
|
|||
token_type_ids=encoded_input["token_type_ids"],
|
||||
attention_mask=encoded_input["attention_mask"])
|
||||
else:
|
||||
from transformers import AutoTokenizer, AutoProcessor
|
||||
text = "Replace me by any text you'd like."
|
||||
if auto_processor is not None and "Tokenizer" not in auto_processor:
|
||||
processor = AutoProcessor.from_pretrained(name)
|
||||
|
|
@ -527,10 +401,11 @@ class TestTransformersModel(TestTorchConvertModel):
|
|||
except:
|
||||
pass
|
||||
if model is None:
|
||||
from transformers import AutoModel
|
||||
model = AutoModel.from_pretrained(name, **model_kwargs)
|
||||
model = self.load_model_with_default_class(name, **model_kwargs)
|
||||
if hasattr(model, "set_default_language"):
|
||||
model.set_default_language("en_XX")
|
||||
if name_suffix != '':
|
||||
model = model._modules[name_suffix]
|
||||
if example is None:
|
||||
if "encodec" in mi.tags:
|
||||
example = (torch.randn(1, 1, 100),)
|
||||
|
|
@ -557,13 +432,25 @@ class TestTransformersModel(TestTorchConvertModel):
|
|||
self.cuda_available, self.gptq_postinit = None, None
|
||||
super().teardown_method()
|
||||
|
||||
@staticmethod
|
||||
def load_model_with_default_class(name, **kwargs):
|
||||
try:
|
||||
mi = model_info(name)
|
||||
assert len({"owlv2", "owlvit", "vit_mae"}.intersection(mi.tags)) == 0, "TBD: support default classes of these models"
|
||||
assert "architectures" in mi.config and len(mi.config["architectures"]) == 1
|
||||
class_name = mi.config["architectures"][0]
|
||||
model_class = transformers.__getattr__(class_name)
|
||||
return model_class.from_pretrained(name, **kwargs)
|
||||
except:
|
||||
return AutoModel.from_pretrained(name, **kwargs)
|
||||
|
||||
@pytest.mark.parametrize("name,type", [("allenai/led-base-16384", "led"),
|
||||
("bert-base-uncased", "bert"),
|
||||
("google/flan-t5-base", "t5"),
|
||||
("google/tapas-large-finetuned-wtq", "tapas"),
|
||||
("gpt2", "gpt2"),
|
||||
("openai/clip-vit-large-patch14", "clip"),
|
||||
("OpenVINO/opt-125m-gptq", "opt")
|
||||
("OpenVINO/opt-125m-gptq", "opt"),
|
||||
])
|
||||
@pytest.mark.precommit
|
||||
def test_convert_model_precommit(self, name, type, ie_device):
|
||||
|
|
|
|||
Loading…
Reference in New Issue