# Set file location and type file_location = "/FileStore/tables/retail_transactions.csv" file_type = "csv" # Define CSV options schema = "orderID INTEGER, customerID INTEGER, productID INTEGER, state STRING, paymentMthd STRING, totalAmt DOUBLE, invoiceTime TIMESTAMP" first_row_is_header = "true" delimiter = "," # Read CSV files into DataFrame df = spark.read.format(file_type) .schema(schema) .option("header", first_row_is_header) .option("delimiter", delimiter) .load(file_location)