messages_turn_1 = [{"role": "user", "content": "What is the capital of France?"}] messages_turn_2 = [{"role": "system", "content": "Today is Tuesday."}, {"role": "user", "content": "What is the capital of France?"}] # tokenize=True is the default and returns a plain list of token ids a = tokenizer.apply_chat_template(messages_turn_1) b = tokenizer.apply_chat_template(messages_turn_2) shared = 0 for x, y in zip(a, b): if x != y: break shared += 1 print(f"shared prefix: {shared} tokens of {len(a)} and {len(b)}") print(f"first divergence at index {shared}: {a[shared:shared+8]} vs {b[shared:shared+8]}") # Output: """ shared prefix: 3 tokens of 35 and 26 diverges at index 3 turn 1: [2683, 418, 253, 11173, 9042, 14260] You are a helpful AI assistant turn 2: [11814, 314, 27758, 30, 2, 198] Today is Tuesday.<|im_end|> """