train_data = session.table("COVID19_RECORDS_PROCESSED") #create train and test dataframes train_data_pd,test_data_pd = train_test_split( train_data_pd, stratify=train_data_pd['TARGET'], test_size=0.1 ) # writing as tempoary tables for mode training and inferencing part session.write_pandas( train_data_pd, table_name="TRAIN_DATA_TMP", auto_create_table=True, table_type="temporary", ) session.write_pandas( test_data_pd, table_name="TEST_DATA_TMP", auto_create_table=True, table_type="temporary", ) train_data_pd = train_data.to_pandas() feature_cols = train_data.columns feature_cols.remove('TARGET') target_col = 'TARGET' model_name = 'decisiontree.model' # How model should be saved in stage model_response = sproc_train_dt_model('TRAIN_DATA_TMP', feature_cols, target_col, model_name, session=session ) print(model_response) # { # "FeatImportance": { # "AGE": 0.4543249401305732, # "ASTHMA": 0.029003830541684678, # "CARDIOVASCULAR": 0.025649097586968667, # "COPD": 0.019300936592021863, # "DIABETES": 0.059273293874405074, # "HIPERTENSION": 0.05885196748765571, # "INMSUPR": 0.0232534703448427, # "INTUBED": 0.026365011429648998, # "MEDICAL_UNIT": 0.08804779552309593, # "OBESITY": 0.02991724846285235, # "OTHER_DISEASE": 0.026840169399286344, # "PATIENT_TYPE": 0, # "PNEUMONIA": 0.04225497414608237, # "PREGNANT": 0.012929499812685114, # "RENAL_CHRONIC": 0.015894267526361774, # "SEX": 0, # "TOBACCO": 0.028563364646896985, # "USMER": 0.059530132494938236 # } # } #plot feature importance feature_coefficients = pd.DataFrame(eval(model_response)) feature_coefficients .sort_values(by='FeatImportance',ascending=False) .plot .bar(y='FeatImportance', figsize=(12,5)) plt.show()