#example sentences sentence1 = 'Hello World!' sentence2 = 'This is an example!' #breaking up into tokens sentence1_tokens = ['[CLS]', 'Hello', 'World', '!'] sentence2_tokens = ['[CLS]', 'This', 'is', 'an', 'example', '!'] #those tokens have IDs sentence1_token_ids = [101, 1340, 87345, 1332] sentence2_token_ids = [101, 4589, 988, 874, 13598, 1332] #we can combine the token ID's together, and add a pad in between them. #also we dont need the CLS token for sentence two sequence_token_ids = [101, 1340, 87345, 1332, 102, 4589, 988, 874, 13598, 1332] #constructing sentence tokens for positional encoding #note how they correspond to the sequence token ids sent_location_tokens=[0, 0, 0, 0, 1, 1, 1, 1, 1, 1 ] #pad, if necessary, with 0's for the tokenids and 1's for the sent locations