from optimum.onnxruntime import ORTModelForSequenceClassification from transformers import pipeline, AutoTokenizer model = ORTModelForSequenceClassification.from_pretrained(onnx_path,file_name="model_optimized_quantized.onnx") tokenizer = AutoTokenizer.from_pretrained(onnx_path) q8_clf = pipeline("text-classification",model=model, tokenizer=tokenizer) q8_clf("What is the exchange rate like on this app?")