while len(ranks) < vocab_size: stats = collections.Counter() for piece in words: for pair in zip(piece[:-1], piece[1:]): stats[pair] += 1 most_common_pair = max(stats, key=lambda x: stats[x]) token_bytes = most_common_pair[0] + most_common_pair[1] token = len(ranks) # Add the new token! ranks[token_bytes] = token # ... then replace that pair through the corpus, and loop