import torch from transformers import AutoModelForVision2Seq, AutoProcessor, BitsAndBytesConfig # Hugging Face model id model_id = "Qwen/Qwen2-VL-7B-Instruct" # BitsAndBytesConfig int-4 config bnb_config = BitsAndBytesConfig( load_in_4bit=True, bnb_4bit_use_double_quant=True, bnb_4bit_quant_type="nf4", bnb_4bit_compute_dtype=torch.bfloat16 ) # Load model and tokenizer model = AutoModelForVision2Seq.from_pretrained( model_id, device_map="auto", # attn_implementation="flash_attention_2", # not supported for training torch_dtype=torch.bfloat16, quantization_config=bnb_config ) processor = AutoProcessor.from_pretrained(model_id)