import torch from transformers import AutoTokenizer, AutoModelForCausalLM, DynamicCache model_id = "HuggingFaceTB/SmolLM2-360M-Instruct" tokenizer = AutoTokenizer.from_pretrained(model_id) model = AutoModelForCausalLM.from_pretrained( model_id, dtype=torch.bfloat16, device_map="auto" ) inputs = tokenizer("The capital of France is", return_tensors="pt") inputs = inputs.to(model.device) past_key_values = DynamicCache(config=model.config) out = model.generate( **inputs, do_sample=False, max_new_tokens=20, past_key_values=past_key_values, ) >>> print(tokenizer.decode(out[0], skip_special_tokens=True)) """The capital of France is Paris. It is the largest city in France and the second-largest city in the European Union.""" >>> print("prompt tokens: ", inputs["input_ids"].shape[1]) "prompt tokens: 5" >>> print("total tokens: ", out.shape[1]) "total tokens: 25" >>> print("cache length: ", past_key_values.get_seq_length()) "cache length: 24"