context = prompt_tokens while not stop: hidden = decoder(context) logits = lm_head(hidden[-1]) probs = softmax(logits / temperature) next_token = decode(probs) context.append(next_token)