context = prompt_tokens while not finished: Q, K, V = compute_qkv(context) output = attention(Q, K, V) next_token = sample(output) context.append(next_token)