def format_dolly_infernece(sample): instruction = f"### Instruction\n{sample['instruction']}" context = f"### Context\n{sample['context']}" if "context" in sample else None response = f"### Answer\n" # join all the parts together prompt = "\n\n".join([i for i in [instruction, context, response] if i is not None]) return prompt def generate(sample): prompt = format_dolly_infernece(sample) inputs = tokenizer(prompt, return_tensors="pt") outputs = model.generate(**inputs, max_new_tokens=512, do_sample=True, temperature=0.9, top_k=50, top_p=0.9) return tokenizer.decode(outputs[0], skip_special_tokens=False)[len(prompt):]