import gradio as gr import pandas as pd import numpy as np import json import plotly.graph_objects as go from typing import Dict, List, Tuple, Optional # Import evaluation modules from evaluator_module import AetherScoreEvaluator from visualizer_module import EvaluationVisualizer from report_generator import ReportGenerator # --- Global Components & Storage --- evaluator = AetherScoreEvaluator() visualizer = EvaluationVisualizer() report_gen = ReportGenerator() # In-memory storage for explainability feature evaluation_storage: Dict[str, Dict] = {} # CSS for better styling custom_css = """ .gradio-container { font-family: 'Inter', sans-serif; } .metric-card { background: linear-gradient(135deg, #667eea 0%, #764ba2 100%); padding: 20px; border-radius: 10px; color: white; margin: 10px 0; } """ # Single Evaluation Process def process_single_evaluation( prompt: str, response: str, expected_answer: Optional[str] = None, agent_name: str = "Agent-1", task_type: str = "general" ) -> Tuple[Dict, go.Figure, go.Figure, str]: # If Prompt or Response is missing from User if not prompt or not response: return {}, go.Figure(), go.Figure(), "Please provide both prompt and response." # Evaluate the response eval_result = evaluator.evaluate_single( prompt=prompt, response=response, expected_answer=expected_answer, task_type=task_type ) scores = eval_result.get("scores", {}) # # Store result for explainability # eval_id = scores.get("eval_id", "") # if eval_id: # evaluation_storage[eval_id] = eval_result # print(f"Stored evaluation {eval_id}") # Generate visualizations spider_chart = visualizer.create_spider_chart(scores, agent_name) score_bars = visualizer.create_score_bars(scores, agent_name) # Generating explanation explanation = evaluator.generate_explanation(scores) # Format scores for display scores_display = {k: f"{v:.2f}" for k, v in scores.items() if isinstance(v, float)} # scores_display["eval_id"] = eval_id return scores_display, spider_chart, score_bars, explanation # For Batch Evaluation def process_batch_evaluation( file_input, evaluation_mode: str = "comprehensive" ) -> Tuple[go.Figure, go.Figure, go.Figure, str, pd.DataFrame]: # If file input is None if file_input is None: return go.Figure(), go.Figure(), go.Figure(), "Please upload a file.", pd.DataFrame() try: if file_input.name.endswith('.json'): with open(file_input.name, 'r') as f: data = json.load(f) elif file_input.name.endswith('.jsonl'): data = [] with open(file_input.name, 'r') as f: for line in f: data.append(json.loads(line)) else: raise ValueError("Unsupported file format. Please upload JSON or JSONL.") # Process batch evaluation results = evaluator.evaluate_batch(data, mode=evaluation_mode) # Store all results for explainability # for res in results: # eval_id = res.get('scores', {}).get('eval_id') # if eval_id: # evaluation_storage[eval_id] = res # Generating visualizations heatmap = visualizer.create_evaluation_heatmap(results) distribution = visualizer.create_score_distribution(results) trends = visualizer.create_performance_trends(results) report = report_gen.generate_batch_report(results) leaderboard = create_leaderboard(results) return heatmap, distribution, trends, report, leaderboard except Exception as e: error_msg = f"Error processing batch file: {str(e)}" return go.Figure(), go.Figure(), go.Figure(), error_msg, pd.DataFrame() def create_leaderboard(results: List[Dict]) -> pd.DataFrame: """Create a leaderboard from evaluation results""" agent_scores = evaluator.get_agent_scores_from_results(results) leaderboard_data = [] for agent, scores in agent_scores.items(): leaderboard_data.append({ 'Rank': 0, 'Agent': agent, 'Avg Score': np.mean(scores), 'Max Score': np.max(scores), 'Min Score': np.min(scores), 'Std Dev': np.std(scores), 'Evaluations': len(scores) }) df = pd.DataFrame(leaderboard_data).sort_values('Avg Score', ascending=False) df['Rank'] = range(1, len(df) + 1) for col in ['Avg Score', 'Max Score', 'Min Score', 'Std Dev']: df[col] = df[col].apply(lambda x: f"{x:.3f}") return df def compare_agents( agent1_file, agent2_file, ) -> Tuple[go.Figure, go.Figure, go.Figure, str]: """Compare two agents' performance""" if not agent1_file or not agent2_file: return go.Figure(), go.Figure(), go.Figure(), "Please upload files for both agents." try: def load_agent_data(file): if file.name.endswith('.json'): with open(file.name, 'r') as f: return json.load(f) elif file.name.endswith('.jsonl'): data = []; with open(file.name, 'r') as f: for line in f: data.append(json.loads(line)) return data raise ValueError("Unsupported file format") agent1_results = evaluator.evaluate_batch(load_agent_data(agent1_file)) agent2_results = evaluator.evaluate_batch(load_agent_data(agent2_file)) comparison_chart = visualizer.create_agent_comparison(agent1_results, agent2_results) radar_comparison = visualizer.create_radar_comparison(agent1_results, agent2_results) performance_delta = visualizer.create_performance_delta(agent1_results, agent2_results) comparison_report = report_gen.generate_comparison_report(agent1_results, agent2_results) return comparison_chart, radar_comparison, performance_delta, comparison_report except Exception as e: error_msg = f"Error comparing agents: {str(e)}" return go.Figure(), go.Figure(), go.Figure(), error_msg def get_detailed_explanation(eval_id: str) -> Tuple[Dict, str]: """Retrieve and format the detailed explanation for an evaluation ID.""" if not eval_id: return {}, "Please enter an Evaluation ID." eval_result = evaluation_storage.get(eval_id) if not eval_result: return {}, f"Error: Evaluation ID '{eval_id}' not found in memory." scores = eval_result.get("scores", {}) reasons = eval_result.get("reasons", {}) # Format details for JSON display details_to_display = { "evaluation_id": scores.get("eval_id"), "timestamp": scores.get("timestamp"), "task_type": scores.get("task_type"), "scores": {k: v for k, v in scores.items() if k not in ["eval_id", "timestamp", "task_type"]} } # Format breakdown for text display breakdown = [] breakdown.append(f"### Detailed Breakdown for Evaluation: {eval_id}\n") for metric, reason in reasons.items(): metric_name = metric.replace('_', ' ').title() score = scores.get(metric, 0) breakdown.append(f"**{metric_name}**: {score:.2f}\n" f"> *Reasoning*: {reason}\n") return details_to_display, "\n".join(breakdown) # --- Gradio Interface --- with gr.Blocks(title="AetherScore - AI Agent Evaluation Framework", css=custom_css) as demo: gr.Markdown("# 🌟 AetherScore - Advanced AI Agent Evaluation Framework") with gr.Tabs(): # Tab 1: Single Evaluation with gr.TabItem("🎯 Single Evaluation"): with gr.Row(): with gr.Column(scale=1): prompt_input = gr.Textbox(label="Prompt", lines=3) response_input = gr.Textbox(label="Agent Response", lines=3) expected_answer = gr.Textbox(label="Expected Answer (Optional)", lines=2) agent_name_input = gr.Textbox(label="Agent Name", value="Agent-1") task_type = gr.Dropdown(label="Task Type", choices=["general", "QA", "summarization", "reasoning", "code"], value="general") evaluate_btn = gr.Button("πŸ” Evaluate", variant="primary") with gr.Column(scale=1): scores_output = gr.JSON(label="πŸ“Š Evaluation Scores") explanation_output = gr.Textbox(label="πŸ“ Explanation", lines=6, interactive=False) with gr.Row(): spider_chart_output = gr.Plot(label="πŸ•ΈοΈ Spider Chart") score_bars_output = gr.Plot(label="πŸ“Š Score Breakdown") evaluate_btn.click(fn=process_single_evaluation, inputs=[prompt_input, response_input, expected_answer, agent_name_input, task_type], outputs=[scores_output, spider_chart_output, score_bars_output, explanation_output]) # Tab 2: Batch Evaluation with gr.TabItem("πŸ“¦ Batch Evaluation"): with gr.Row(): with gr.Column(scale=1): batch_file = gr.File(label="Upload Evaluation File (JSON/JSONL)", file_types=[".json", ".jsonl"]) evaluation_mode = gr.Radio(label="Evaluation Mode", choices=["comprehensive", "fast", "detailed"], value="comprehensive") batch_evaluate_btn = gr.Button("πŸš€ Run Batch Evaluation", variant="primary") with gr.Column(scale=2): batch_report = gr.Textbox(label="πŸ“„ Evaluation Report", lines=10, interactive=False) with gr.Row(): heatmap_output = gr.Plot(label="πŸ—ΊοΈ Evaluation Heatmap") distribution_output = gr.Plot(label="πŸ“ˆ Score Distribution") trends_output = gr.Plot(label="πŸ“Š Performance Trends") leaderboard_output = gr.Dataframe(label="πŸ† Agent Leaderboard", interactive=False) batch_evaluate_btn.click(fn=process_batch_evaluation, inputs=[batch_file, evaluation_mode], outputs=[heatmap_output, distribution_output, trends_output, batch_report, leaderboard_output]) # Tab 3: Agent Comparison with gr.TabItem("βš”οΈ Agent Comparison"): with gr.Row(): with gr.Column(): agent1_file = gr.File(label="Agent 1 Results (JSON/JSONL)", file_types=[".json", ".jsonl"]) agent2_file = gr.File(label="Agent 2 Results (JSON/JSONL)", file_types=[".json", ".jsonl"]) compare_btn = gr.Button("βš–οΈ Compare Agents", variant="primary") with gr.Column(): comparison_report = gr.Textbox(label="πŸ“Š Comparison Analysis", lines=8, interactive=False) with gr.Row(): comparison_chart = gr.Plot(label="πŸ“Š Performance Comparison") radar_comparison = gr.Plot(label="🎯 Radar Comparison") performance_delta = gr.Plot(label="πŸ“ˆ Performance Delta Analysis") compare_btn.click(fn=compare_agents, inputs=[agent1_file, agent2_file], outputs=[comparison_chart, radar_comparison, performance_delta, comparison_report]) # Tab 4: Explainability Dashboard with gr.TabItem("πŸ” Explainability"): gr.Markdown("### Deep dive into evaluation decisions. Enter an `eval_id` from a previous run.") with gr.Row(): eval_id_input = gr.Textbox(label="Evaluation ID", placeholder="Enter evaluation ID to analyze...") load_eval_btn = gr.Button("Load Evaluation", variant="secondary") with gr.Row(): eval_details = gr.JSON(label="Evaluation Details") detailed_breakdown = gr.Markdown(label="Detailed Score Breakdown") load_eval_btn.click(fn=get_detailed_explanation, inputs=[eval_id_input], outputs=[eval_details, detailed_breakdown]) if __name__ == "__main__": demo.launch(share=True)