import torch from transformers import GPT2LMHeadModel, GPT2Config torch.set_float32_matmul_precision('high') DEVICE = "cuda" # define the decoder model config = GPT2Config.from_pretrained("gpt2") model = GPT2LMHeadModel(config).to(DEVICE).eval() @torch.inference_mode() def generate_sequence(model, max_seqlen, batch_size): # Initialize prompts with BOS token all_tokens = torch.full( (batch_size, 1), config.bos_token_id, device=DEVICE, dtype=torch.long ) finished = torch.zeros(batch_size, device=DEVICE, dtype=torch.bool) for i in range(max_seqlen): outputs = model(all_tokens) # extract new token logits = outputs.logits[:, -1, :] new_tokens = torch.argmax(logits, dim=-1) # append new token to sequence all_tokens = torch.cat( [all_tokens, new_tokens.unsqueeze(-1)], dim=-1 ) finished |= (new_tokens == config.eos_token_id) stop_gpu = torch.all(finished) # checking stop condition if stop_gpu.item(): print(f"All sequences finished at step {i+1}") break return all_tokens