import torch from transformers import AutoModelForCausalLM, AutoTokenizer # Hugging Face model id model_id = "philschmid/llama-2-7b-instruction-generator" tokenizer = AutoTokenizer.from_pretrained(model_id, use_fast=False) # bug with fast tokenizer model = AutoModelForCausalLM.from_pretrained(model_id, low_cpu_mem_usage=True, torch_dtype=torch.float16) # we load the model in fp16 on purpose