from anthropic_tokenizers import TiktokenBPE # Define model identifiers models = { "Claude 3 Opus": "claude-3-opus-20240229", "Claude 3 Sonnet": "claude-3-sonnet-20240229", "Claude 3 Haiku": "claude-3-haiku-20240307", } # Sample text for comparison comparison_text = "This is a sentence designed to test tokenization consistency across Claude 3 models. It includes punctuation! and numbers 12345. It also has some longer words like 'tokenization' and 'consistency'." print(f"--- Comparing Token Counts Across Claude 3 Models ---") print(f"Text for comparison:\n'{comparison_text}'\n") for model_name, model_id in models.items(): try: # The TiktokenBPE class in anthropic-tokenizers uses tiktoken, # which maps these model names to specific encodings. # For Claude 3 family, they all map to 'cl100k_base'. tokenizer = TiktokenBPE(model_id) num_tokens = tokenizer.count_tokens(comparison_text) print(f"{model_name} ({model_id}): {num_tokens} tokens (Encoding: {tokenizer.encoding_name})") except ValueError as e: print(f"Could not initialize tokenizer for {model_name} ({model_id}): {e}") # If a specific model ID fails, it might be due to library updates or mapping. # We can try the common encoder name directly if this happens. try: tokenizer = TiktokenBPE("cl100k_base") # The common encoder for Claude 3 num_tokens = tokenizer.count_tokens(comparison_text) print(f" -> Fallback using 'cl100k_base': {num_tokens} tokens (Encoding: {tokenizer.encoding_name})") except Exception as fallback_e: print(f" -> Fallback failed: {fallback_e}")