from anthropic_tokenizers import TiktokenBPE # Initialize a tokenizer instance. # For Claude 3 models, the underlying tokenizer is generally consistent. # However, specifying the model name is good practice. # Let's assume a generic Claude 3 tokenizer for demonstration, # as specific model variations in tokenization are not publicly documented to be significant enough # to warrant different tokenizer instances in the provided library for Claude 3. # If future models introduce divergence, this would be the place to specify it. # Based on documentation and common practice, it's often a single tokenizer # for a family of models, or slight variations. Let's use a representative one. # The library abstracts this. For Claude 3, we can instantiate it. # Note: The anthropic-tokenizers library primarily relies on the tiktoken encoder, # which is generally consistent across a model family unless explicitly stated otherwise. # For practical purposes of Claude 3 family (Opus, Sonnet, Haiku), the underlying # BPE encoding is typically the same. try: tokenizer_claude_3 = TiktokenBPE("claude-3-opus-20240229") except ValueError: # Fallback or error handling if the specific model name isn't directly supported # In practice, for Claude 3, the encoding is often shared. # Let's try a common alias or a base if the specific version isn't found. # The library might dynamically map these. print("Specific model name not found directly, attempting a common encoder.") # This part is illustrative; the library handles mappings. # For Claude 3, `claude-3-opus-20240229`, `claude-3-sonnet-20240229`, and `claude-3-haiku-20240307` # all use the same underlying `cl100k_base` encoding scheme found in OpenAI's GPT-4. # The `anthropic-tokenizers` library abstracts this. # Let's instantiate using a known encoder name that Anthropic uses internally for Claude 3. # The library might abstract this into a single `Claude3Tokenizer` class or similar. # However, based on the `anthropic-tokenizers` source and usage patterns, it directly maps # to `tiktoken` encoders. The common encoder for Claude 3 models is `cl100k_base`. tokenizer_claude_3 = TiktokenBPE("cl100k_base") # This is the underlying encoder. print(f"Tokenizer initialized. Encoding: {tokenizer_claude_3.encoding_name}") # Let's define some sample texts to analyze. text_short = "Hello, world!" text_sentence = "The quick brown fox jumps over the lazy dog." text_paragraph = """ Tokenization is the process of breaking down a sequence of text into smaller units, called tokens. These tokens can be words, sub-words, or even individual characters. The way text is tokenized can have a significant impact on the performance and cost of large language models. Understanding token counts is crucial for managing context windows and API usage. """ text_code = """ def greet(name): return f"Hello, {name}!" print(greet("Alice")) """ text_special_chars = "This is a test with some special characters: !@#$%^&*()_+=-`~[]{}|;:'\",.<>/? and numbers 12345." text_english_chinese = "Hello, 你好世界!"