#!/usr/bin/env python3 """ Merge autonomy scores from ai_involvement_checklist_responses.csv into papers.csv """ import csv import re from difflib import SequenceMatcher def normalize_title(title): """Normalize title for matching""" # Remove special characters and convert to lowercase title = re.sub(r'[^\w\s]', '', title.lower()) # Remove extra whitespace title = ' '.join(title.split()) return title def extract_title_from_filename(filename): """Extract title from filename like '312_Superior Energy Storage...pdf'""" # Remove number prefix and .pdf suffix title = re.sub(r'^\d+_', '', filename) title = re.sub(r'\.pdf$', '', title) # Handle truncated titles (ending with ...) title = title.replace('...', '') return normalize_title(title) def similarity(a, b): """Calculate similarity ratio between two strings""" return SequenceMatcher(None, a, b).ratio() def find_best_match(target_title, paper_titles): """Find the best matching paper title""" target_norm = normalize_title(target_title) best_match = None best_score = 0 for paper_title in paper_titles: paper_norm = normalize_title(paper_title) # Check if one is a prefix of the other (for truncated titles) if target_norm.startswith(paper_norm) or paper_norm.startswith(target_norm): score = 0.95 else: score = similarity(target_norm, paper_norm) if score > best_score: best_score = score best_match = paper_title # Only return match if similarity is high enough if best_score > 0.7: return best_match, best_score return None, 0 # Read papers.csv papers = {} with open('papers.csv', 'r', encoding='utf-8') as f: reader = csv.DictReader(f) for row in reader: papers[row['title']] = row # Read autonomy data autonomy_data = {} with open('ai_involvement_checklist_responses.csv', 'r', encoding='utf-8') as f: reader = csv.DictReader(f) for row in reader: if row['paper_name'] and row['checklist_found'].upper() == 'TRUE': paper_name = row['paper_name'] extracted_title = extract_title_from_filename(paper_name) # Find matching paper match, score = find_best_match(extracted_title, papers.keys()) if match: autonomy_data[match] = { 'hypothesis_development': row['hypothesis_development_answer'], 'experimental_design': row['experimental_design_answer'], 'data_analysis': row['data_analysis_answer'], 'writing': row['writing_answer'] } print(f"Matched: {paper_name} -> {match[:50]}... (score: {score:.2f})") else: print(f"No match found for: {paper_name} (extracted: {extracted_title[:50]}...)") print(f"\nMatched {len(autonomy_data)} papers with autonomy data") # Write merged CSV with open('papers.csv', 'w', encoding='utf-8', newline='') as f: fieldnames = ['title', 'link', 'AIRev1_score', 'AIRev2_score', 'AIRev3_score', 'Human_score', 'status', 'hypothesis_development', 'experimental_design', 'data_analysis', 'writing', 'primary_topic', 'secondary_topic'] writer = csv.DictWriter(f, fieldnames=fieldnames) writer.writeheader() for title, row in papers.items(): output_row = { 'title': row['title'], 'link': row['link'], 'AIRev1_score': row['AIRev1_score'], 'AIRev2_score': row['AIRev2_score'], 'AIRev3_score': row['AIRev3_score'], 'Human_score': row['Human_score'], 'status': row['status'], 'hypothesis_development': row.get('hypothesis_development', ''), 'experimental_design': row.get('experimental_design', ''), 'data_analysis': row.get('data_analysis', ''), 'writing': row.get('writing', ''), 'primary_topic': row.get('primary_topic', ''), 'secondary_topic': row.get('secondary_topic', '') } if title in autonomy_data: # Update with new autonomy data output_row['hypothesis_development'] = autonomy_data[title]['hypothesis_development'] output_row['experimental_design'] = autonomy_data[title]['experimental_design'] output_row['data_analysis'] = autonomy_data[title]['data_analysis'] output_row['writing'] = autonomy_data[title]['writing'] writer.writerow(output_row) print(f"\nWrote merged data to papers.csv")