import torch from transformers import AutoTokenizer, AutoModelForCausalLM, DynamicCache model_id = "HuggingFaceTB/SmolLM2-360M-Instruct" tokenizer = AutoTokenizer.from_pretrained(model_id) model = AutoModelForCausalLM.from_pretrained( model_id, dtype=torch.bfloat16, device_map="auto" ) past_key_values = DynamicCache(config=model.config) messages = [] questions = ["What is the capital of France?", "And its population?"] for prompt in questions: # Add to the history messages.append({"role": "user", "content": prompt}) # Tokenize inputs = tokenizer.apply_chat_template( messages, add_generation_prompt=True, return_tensors="pt", return_dict=True ).to(model.device) # Generate input_length = inputs["input_ids"].shape[1] outputs = model.generate( **inputs, do_sample=False, max_new_tokens=64, past_key_values=past_key_values ) # decode completion = tokenizer.decode(outputs[0, input_length:], skip_special_tokens=True) # Append to message history messages.append({"role": "assistant", "content": completion}) print(f"turn tokens in: {input_length} | cache now: {past_key_values.get_seq_length()}") # Output: "turn tokens in: 42 | cache now: 55" "turn tokens in: 71 | cache now: 92"