import time # The prompt is based on the fine-tuning from the model: https://www.philschmid.de/instruction-tune-llama-2#4-test-model-and-run-inference prompt = """### Instruction: Use the Input below to create an instruction, which could have been used to generate the input using an LLM. ### Input: Dear [boss name], I'm writing to request next week, August 1st through August 4th, off as paid time off. I have some personal matters to attend to that week that require me to be out of the office. I wanted to give you as much advance notice as possible so you can plan accordingly while I am away. Thank you, [Your name] ### Response: """ # helper function to generate text and measure latency def generate_helper(pipeline,prompt=prompt): # warm up for i in range(5): _ = pipeline("Warm up") # measure latency in a simple way start = time.time() out = pipeline(prompt, max_new_tokens=100, do_sample=True, top_p=0.9,temperature=0.9) end = time.time() generated_text = out[0]["generated_text"][len(prompt):] latency_per_token_in_ms = ((end-start)/len(pipeline.tokenizer(generated_text)["input_ids"]))*1000 # return the generated text and the latency return {"text": out[0]["generated_text"][len(prompt):], "latency": f"{round(latency_per_token_in_ms,2)}ms/token"}