import pandas as pd import matplotlib.pyplot as plt import seaborn as sns from collections import Counter # Read the CSV files agent_use = pd.read_csv('agent_use.csv') papers = pd.read_csv('papers_with_ids.csv') # Rename Paper ID column for consistency agent_use = agent_use.rename(columns={'Paper ID': 'paper_id'}) # Clean and ensure paper_id is string type for merging # Remove any NaN values and convert to string agent_use = agent_use[agent_use['paper_id'].notna()].copy() papers = papers[papers['paper_id'].notna()].copy() # Convert to int first to remove decimals, then to string agent_use['paper_id'] = agent_use['paper_id'].astype(float).astype(int).astype(str).str.strip() papers['paper_id'] = papers['paper_id'].astype(float).astype(int).astype(str).str.strip() # Merge agent_use with papers to get status merged = agent_use.merge(papers[['paper_id', 'status']], on='paper_id', how='left') print(f"Total agent use entries: {len(agent_use)}") print(f"Entries with status after merge: {merged['status'].notna().sum()}") # Parse the Agent/Model/Platform column - it may contain multiple agents separated by commas def parse_agents(agent_string): if pd.isna(agent_string): return [] # Split by comma and clean up whitespace agents = [a.strip() for a in str(agent_string).split(',')] return agents # Define broad terms to exclude broad_terms = ["AI", "AI agent", "AI systems", "LLM", "AI tools", "LLM assistant", "AI agents", "AI system", "AI partner", "AI pipeline", "AI collaborator", "LLMs"] # Count agent usage for accepted papers only accepted_merged = merged[merged['status'] == 'accepted'] accepted_agents = [] for agent_string in accepted_merged['Agent/Model/Platform']: agents = parse_agents(agent_string) # Filter out broad terms agents = [a for a in agents if a not in broad_terms] accepted_agents.extend(agents) # Get frequency counts accepted_agent_counts = Counter(accepted_agents) print(f"\nTotal accepted papers using agents: {len(accepted_merged)}") print(f"Unique agents in accepted papers: {len(accepted_agent_counts)}") # Convert to DataFrame for plotting (sort by count descending for display) accepted_df = pd.DataFrame(list(accepted_agent_counts.items()), columns=['Agent', 'Count']).sort_values('Count', ascending=False) # Create the plot fig, ax = plt.subplots(figsize=(14, 6)) # Vertical bar plot ax.bar(accepted_df['Agent'], accepted_df['Count'], color='forestgreen', alpha=0.8) ax.set_ylabel('Number of Papers', fontsize=12) ax.set_xlabel('Agent/Model/Platform', fontsize=12) ax.set_title('AI Agent/Model/Platform Usage in Accepted Papers', fontsize=14, fontweight='bold') ax.grid(axis='y', alpha=0.3) ax.set_ylim(bottom=0, top=4.5) # Rotate x-axis labels to 45 degrees plt.xticks(rotation=45, ha='right') # Add value labels on top of bars for i, (agent, count) in enumerate(zip(accepted_df['Agent'], accepted_df['Count'])): ax.text(i, count + 0.1, str(count), ha='center', va='bottom', fontsize=9) plt.tight_layout() plt.savefig('agent_usage_comparison.png', dpi=300, bbox_inches='tight') print("\nPlot saved as: agent_usage_comparison.png") # Print summary statistics print("\n" + "="*60) print("SUMMARY STATISTICS") print("="*60) print(f"\nAccepted Papers:") print(accepted_df.to_string(index=False)) print(f"\nBroad terms excluded: {broad_terms}")