# Remove the columns papers = papers.drop(columns=['id', 'event_type', 'pdf_name'], axis=1).sample(100) # Print out the first rows of papers papers.head()