# Function to read the first page of a PDF and extract the abstract def extract_abstract(pdf_path): # Open the PDF file and grab text from the 1st page with fitz.open(pdf_path) as pdf: first_page = pdf[0] text = first_page.get_text("text") # Extract the abstract (assuming the abstract starts with 'Abstract') # find where abstract starts start_idx = text.lower().find('abstract') # end abstract at introduction if it exists on 1st page if 'introduction' in text.lower(): end_idx = text.lower().find('introduction') else: end_idx = None # extract abstract text abstract = text[start_idx:end_idx].strip() # if abstract appears on 1st page return it, if not resturn None if start_idx != -1: abstract = text[start_idx:end_idx].strip() return abstract else: return None