# report_generator.py from typing import Dict, List import numpy as np from collections import defaultdict from datetime import datetime class ReportGenerator: """ Generates textual and HTML reports from evaluation results """ def generate_batch_report(self, results: List[Dict]) -> str: """ Generates a comprehensive text report for a batch evaluation. Args: results: A list of evaluation result dictionaries. Returns: A formatted string containing the batch evaluation report. """ if not results: return "No evaluation results to report." num_evals = len(results) agent_names = list(set(r['agent_name'] for r in results)) num_agents = len(agent_names) # Aggregate scores overall_scores = [r['scores']['overall_score'] for r in results] metric_scores = defaultdict(list) agent_overall_scores = defaultdict(list) for res in results: agent_overall_scores[res['agent_name']].append(res['scores']['overall_score']) for metric, score in res['scores'].items(): if metric not in ['eval_id', 'timestamp', 'task_type']: metric_scores[metric].append(score) # Calculate agent averages agent_avg_scores = {agent: np.mean(scores) for agent, scores in agent_overall_scores.items()} top_agent = max(agent_avg_scores, key=agent_avg_scores.get) bottom_agent = min(agent_avg_scores, key=agent_avg_scores.get) # Build report string report = [] report.append("="*50) report.append(" AetherScore - Batch Evaluation Report") report.append("="*50) report.append(f"Report generated on: {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}\n") report.append("--- Summary ---") report.append(f"Total Evaluations: {num_evals}") report.append(f"Number of Agents: {num_agents}") report.append(f"Overall Average Score: {np.mean(overall_scores):.3f}\n") report.append("--- Agent Performance ---") report.append(f"Top Performing Agent: {top_agent} (Avg Score: {agent_avg_scores[top_agent]:.3f})") report.append(f"Agent with most room for improvement: {bottom_agent} (Avg Score: {agent_avg_scores[bottom_agent]:.3f})\n") report.append("--- Metric Breakdown (Average Scores) ---") for metric, scores in metric_scores.items(): metric_name = metric.replace('_', ' ').title() report.append(f"- {metric_name:<25}: {np.mean(scores):.3f}") report.append("\n" + "="*50) return "\n".join(report) def generate_comparison_report( self, agent1_results: List[Dict], agent2_results: List[Dict] ) -> str: """ Generates a text report comparing two agents. Args: agent1_results: Evaluation results for the first agent. agent2_results: Evaluation results for the second agent. Returns: A formatted string comparing the two agents. """ if not agent1_results or not agent2_results: return "Insufficient data for comparison. Please provide results for both agents." agent1_name = agent1_results[0].get('agent_name', 'Agent 1') agent2_name = agent2_results[0].get('agent_name', 'Agent 2') # Calculate average scores for each agent metrics = ['overall_score', 'instruction_following', 'hallucination_score', 'assumption_control', 'coherence', 'accuracy'] avg_scores1 = {m: np.mean([r['scores'].get(m, 0) for r in agent1_results]) for m in metrics} avg_scores2 = {m: np.mean([r['scores'].get(m, 0) for r in agent2_results]) for m in metrics} # Build report string report = [] report.append("="*60) report.append(f" Agent Comparison Report: {agent1_name} vs. {agent2_name}") report.append("="*60) report.append(f"Report generated on: {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}\n") # Overall Winner winner = agent1_name if avg_scores1['overall_score'] > avg_scores2['overall_score'] else agent2_name report.append("--- Overall Performance ---") report.append(f"🏆 Winner: {winner}") report.append(f"{agent1_name} Avg Overall Score: {avg_scores1['overall_score']:.3f}") report.append(f"{agent2_name} Avg Overall Score: {avg_scores2['overall_score']:.3f}\n") report.append("--- Detailed Metric Comparison ---") header = f"{'Metric':<25} | {agent1_name:<10} | {agent2_name:<10} | {'Delta':<8} | {'Winner'}" report.append(header) report.append("-"*len(header)) for metric in metrics: s1 = avg_scores1[metric] s2 = avg_scores2[metric] delta = s2 - s1 metric_winner = agent1_name if s1 > s2 else agent2_name if s2 > s1 else "Tie" metric_name = metric.replace('_', ' ').title() report.append(f"{metric_name:<25} | {s1:<10.3f} | {s2:<10.3f} | {delta:<+8.3f} | {metric_winner}") report.append("\n" + "="*60) return "\n".join(report) def generate_html_report(self, results_data: List[Dict]) -> str: """ Generates a basic HTML report from evaluation results. Args: results_data: A list of evaluation result dictionaries. Returns: A string containing a full HTML report. """ report_str = self.generate_batch_report(results_data) # Basic HTML template html_template = f""" AetherScore Evaluation Report

AetherScore Evaluation Report

This report contains a summary of the batch evaluation results.

{report_str}

Note: This is a text-based summary. For interactive visualizations, please use the AetherScore dashboard.

""" return html_template