def get_summary_context_message(df: pd.DataFrame) -> str: """ Generate a comprehensive summary report of MBA admissions dataset statistics. This function analyzes MBA application data to provide detailed statistics on applicant demographics, academic performance, professional backgrounds, and admission rates across various categories. The summary includes gender and international status distributions, GPA and GMAT score statistics, admission rates by academic major and work industry, and work experience impact analysis. Parameters ---------- df : pd.DataFrame DataFrame containing MBA admissions data with the following expected columns: - 'gender', 'international', 'gpa', 'gmat', 'major', 'work_industry', 'work_exp', 'admission' Returns ------- str A formatted multi-line string containing comprehensive MBA admissions statistics. """ # Basic application statistics total_applications = len(df) # Gender distribution gender_counts = df["gender"].value_counts() male_count = gender_counts.get("Male", 0) female_count = gender_counts.get("Female", 0) # International status international_count = ( df["international"].sum() if df["international"].dtype == bool else (df["international"] == True).sum() ) # GPA statistics gpa_data = df["gpa"].dropna() gpa_avg = gpa_data.mean() gpa_25th = gpa_data.quantile(0.25) gpa_50th = gpa_data.quantile(0.50) gpa_75th = gpa_data.quantile(0.75) # GMAT statistics gmat_data = df["gmat"].dropna() gmat_avg = gmat_data.mean() gmat_25th = gmat_data.quantile(0.25) gmat_50th = gmat_data.quantile(0.50) gmat_75th = gmat_data.quantile(0.75) # Major analysis - admission rates by major major_stats = [] for major in df["major"].unique(): major_data = df[df["major"] == major] admitted = len(major_data[major_data["admission"] == "Admit"]) total = len(major_data) rate = (admitted / total) * 100 major_stats.append((major, admitted, total, rate)) # Sort by admission rate (descending) major_stats.sort(key=lambda x: x[3], reverse=True) # Work industry analysis - admission rates by industry industry_stats = [] for industry in df["work_industry"].unique(): if pd.isna(industry): continue industry_data = df[df["work_industry"] == industry] admitted = len(industry_data[industry_data["admission"] == "Admit"]) total = len(industry_data) rate = (admitted / total) * 100 industry_stats.append((industry, admitted, total, rate)) # Sort by admission rate (descending) industry_stats.sort(key=lambda x: x[3], reverse=True) # Work experience analysis work_exp_data = df["work_exp"].dropna() avg_work_exp_all = work_exp_data.mean() # Work experience for admitted students admitted_students = df[df["admission"] == "Admit"] admitted_work_exp = admitted_students["work_exp"].dropna() avg_work_exp_admitted = admitted_work_exp.mean() # Work experience ranges analysis def categorize_work_exp(exp): if pd.isna(exp): return "Unknown" elif exp < 2: return "0-1 years" elif exp < 4: return "2-3 years" elif exp < 6: return "4-5 years" elif exp < 8: return "6-7 years" else: return "8+ years" df["work_exp_category"] = df["work_exp"].apply(categorize_work_exp) work_exp_category_stats = [] for category in ["0-1 years", "2-3 years", "4-5 years", "6-7 years", "8+ years"]: category_data = df[df["work_exp_category"] == category] if len(category_data) > 0: admitted = len(category_data[category_data["admission"] == "Admit"]) total = len(category_data) rate = (admitted / total) * 100 work_exp_category_stats.append((category, admitted, total, rate)) # Build the summary message summary = f"""MBA Admissions Dataset Summary (2025) Total Applications: {total_applications:,} people applied to the MBA program. Gender Distribution: - Male applicants: {male_count:,} ({male_count/total_applications*100:.1f}%) - Female applicants: {female_count:,} ({female_count/total_applications*100:.1f}%) International Status: - International applicants: {international_count:,} ({international_count/total_applications*100:.1f}%) - Domestic applicants: {total_applications-international_count:,} ({(total_applications-international_count)/total_applications*100:.1f}%) Academic Performance Statistics: GPA Statistics: - Average GPA: {gpa_avg:.2f} - 25th percentile: {gpa_25th:.2f} - 50th percentile (median): {gpa_50th:.2f} - 75th percentile: {gpa_75th:.2f} GMAT Statistics: - Average GMAT: {gmat_avg:.0f} - 25th percentile: {gmat_25th:.0f} - 50th percentile (median): {gmat_50th:.0f} - 75th percentile: {gmat_75th:.0f} Major Analysis - Admission Rates by Academic Background:""" for major, admitted, total, rate in major_stats: summary += ( f"\n- {major}: {admitted}/{total} admitted ({rate:.1f}% admission rate)" ) summary += ( "\n\nWork Industry Analysis - Admission Rates by Professional Background:" ) # Show top 8 industries by admission rate for industry, admitted, total, rate in industry_stats[:8]: summary += ( f"\n- {industry}: {admitted}/{total} admitted ({rate:.1f}% admission rate)" ) summary += "\n\nWork Experience Impact on Admissions:\n\nOverall Work Experience Comparison:" summary += ( f"\n- Average work experience (all applicants): {avg_work_exp_all:.1f} years" ) summary += f"\n- Average work experience (admitted students): {avg_work_exp_admitted:.1f} years" summary += "\n\nAdmission Rates by Work Experience Range:" for category, admitted, total, rate in work_exp_category_stats: summary += ( f"\n- {category}: {admitted}/{total} admitted ({rate:.1f}% admission rate)" ) # Key insights best_major = major_stats[0] best_industry = industry_stats[0] summary += "\n\nKey Insights:" summary += ( f"\n- Highest admission rate by major: {best_major[0]} at {best_major[3]:.1f}%" ) summary += f"\n- Highest admission rate by industry: {best_industry[0]} at {best_industry[3]:.1f}%" if avg_work_exp_admitted > avg_work_exp_all: summary += f"\n- Admitted students have slightly more work experience on average ({avg_work_exp_admitted:.1f} vs {avg_work_exp_all:.1f} years)" else: summary += "\n- Work experience shows minimal difference between admitted and all applicants" return summary