e6test / app.py
aaditya-raj's picture
Upload app.py
0701672 verified
Raw History Blame
12 kB
import gradio as gr
import pandas as pd
import numpy as np
import json
import plotly.graph_objects as go
from typing import Dict, List, Tuple, Optional
# Import evaluation modules
from evaluator_module import AetherScoreEvaluator
from visualizer_module import EvaluationVisualizer
from report_generator import ReportGenerator
# --- Global Components & Storage ---
evaluator = AetherScoreEvaluator()
visualizer = EvaluationVisualizer()
report_gen = ReportGenerator()
# In-memory storage for explainability feature
evaluation_storage: Dict[str, Dict] = {}
# CSS for better styling
custom_css = """
.gradio-container {
font-family: 'Inter', sans-serif;
}
.metric-card {
background: linear-gradient(135deg, #667eea 0%, #764ba2 100%);
padding: 20px;
border-radius: 10px;
color: white;
margin: 10px 0;
}
"""
# Single Evaluation Process
def process_single_evaluation(
prompt: str,
response: str,
expected_answer: Optional[str] = None,
agent_name: str = "Agent-1",
task_type: str = "general"
) -> Tuple[Dict, go.Figure, go.Figure, str]:
# If Prompt or Response is missing from User
if not prompt or not response:
return {}, go.Figure(), go.Figure(), "Please provide both prompt and response."
# Evaluate the response
eval_result = evaluator.evaluate_single(
prompt=prompt,
response=response,
expected_answer=expected_answer,
task_type=task_type
)
scores = eval_result.get("scores", {})
# # Store result for explainability
# eval_id = scores.get("eval_id", "")
# if eval_id:
# evaluation_storage[eval_id] = eval_result
# print(f"Stored evaluation {eval_id}")
# Generate visualizations
spider_chart = visualizer.create_spider_chart(scores, agent_name)
score_bars = visualizer.create_score_bars(scores, agent_name)
# Generating explanation
explanation = evaluator.generate_explanation(scores)
# Format scores for display
scores_display = {k: f"{v:.2f}" for k, v in scores.items() if isinstance(v, float)}
# scores_display["eval_id"] = eval_id
return scores_display, spider_chart, score_bars, explanation
# For Batch Evaluation
def process_batch_evaluation(
file_input,
evaluation_mode: str = "comprehensive"
) -> Tuple[go.Figure, go.Figure, go.Figure, str, pd.DataFrame]:
# If file input is None
if file_input is None:
return go.Figure(), go.Figure(), go.Figure(), "Please upload a file.", pd.DataFrame()
try:
if file_input.name.endswith('.json'):
with open(file_input.name, 'r') as f:
data = json.load(f)
elif file_input.name.endswith('.jsonl'):
data = []
with open(file_input.name, 'r') as f:
for line in f:
data.append(json.loads(line))
else:
raise ValueError("Unsupported file format. Please upload JSON or JSONL.")
# Process batch evaluation
results = evaluator.evaluate_batch(data, mode=evaluation_mode)
# Store all results for explainability
# for res in results:
# eval_id = res.get('scores', {}).get('eval_id')
# if eval_id:
# evaluation_storage[eval_id] = res
# Generating visualizations
heatmap = visualizer.create_evaluation_heatmap(results)
distribution = visualizer.create_score_distribution(results)
trends = visualizer.create_performance_trends(results)
report = report_gen.generate_batch_report(results)
leaderboard = create_leaderboard(results)
return heatmap, distribution, trends, report, leaderboard
except Exception as e:
error_msg = f"Error processing batch file: {str(e)}"
return go.Figure(), go.Figure(), go.Figure(), error_msg, pd.DataFrame()
def create_leaderboard(results: List[Dict]) -> pd.DataFrame:
"""Create a leaderboard from evaluation results"""
agent_scores = evaluator.get_agent_scores_from_results(results)
leaderboard_data = []
for agent, scores in agent_scores.items():
leaderboard_data.append({
'Rank': 0, 'Agent': agent, 'Avg Score': np.mean(scores),
'Max Score': np.max(scores), 'Min Score': np.min(scores),
'Std Dev': np.std(scores), 'Evaluations': len(scores)
})
df = pd.DataFrame(leaderboard_data).sort_values('Avg Score', ascending=False)
df['Rank'] = range(1, len(df) + 1)
for col in ['Avg Score', 'Max Score', 'Min Score', 'Std Dev']:
df[col] = df[col].apply(lambda x: f"{x:.3f}")
return df
def compare_agents(
agent1_file,
agent2_file,
) -> Tuple[go.Figure, go.Figure, go.Figure, str]:
"""Compare two agents' performance"""
if not agent1_file or not agent2_file:
return go.Figure(), go.Figure(), go.Figure(), "Please upload files for both agents."
try:
def load_agent_data(file):
if file.name.endswith('.json'):
with open(file.name, 'r') as f: return json.load(f)
elif file.name.endswith('.jsonl'):
data = [];
with open(file.name, 'r') as f:
for line in f: data.append(json.loads(line))
return data
raise ValueError("Unsupported file format")
agent1_results = evaluator.evaluate_batch(load_agent_data(agent1_file))
agent2_results = evaluator.evaluate_batch(load_agent_data(agent2_file))
comparison_chart = visualizer.create_agent_comparison(agent1_results, agent2_results)
radar_comparison = visualizer.create_radar_comparison(agent1_results, agent2_results)
performance_delta = visualizer.create_performance_delta(agent1_results, agent2_results)
comparison_report = report_gen.generate_comparison_report(agent1_results, agent2_results)
return comparison_chart, radar_comparison, performance_delta, comparison_report
except Exception as e:
error_msg = f"Error comparing agents: {str(e)}"
return go.Figure(), go.Figure(), go.Figure(), error_msg
def get_detailed_explanation(eval_id: str) -> Tuple[Dict, str]:
"""Retrieve and format the detailed explanation for an evaluation ID."""
if not eval_id:
return {}, "Please enter an Evaluation ID."
eval_result = evaluation_storage.get(eval_id)
if not eval_result:
return {}, f"Error: Evaluation ID '{eval_id}' not found in memory."
scores = eval_result.get("scores", {})
reasons = eval_result.get("reasons", {})
# Format details for JSON display
details_to_display = {
"evaluation_id": scores.get("eval_id"),
"timestamp": scores.get("timestamp"),
"task_type": scores.get("task_type"),
"scores": {k: v for k, v in scores.items() if k not in ["eval_id", "timestamp", "task_type"]}
}
# Format breakdown for text display
breakdown = []
breakdown.append(f"### Detailed Breakdown for Evaluation: {eval_id}\n")
for metric, reason in reasons.items():
metric_name = metric.replace('_', ' ').title()
score = scores.get(metric, 0)
breakdown.append(f"**{metric_name}**: {score:.2f}\n"
f"> *Reasoning*: {reason}\n")
return details_to_display, "\n".join(breakdown)
# --- Gradio Interface ---
with gr.Blocks(title="AetherScore - AI Agent Evaluation Framework", css=custom_css) as demo:
gr.Markdown("# 🌟 AetherScore - Advanced AI Agent Evaluation Framework")
with gr.Tabs():
# Tab 1: Single Evaluation
with gr.TabItem("🎯 Single Evaluation"):
with gr.Row():
with gr.Column(scale=1):
prompt_input = gr.Textbox(label="Prompt", lines=3)
response_input = gr.Textbox(label="Agent Response", lines=3)
expected_answer = gr.Textbox(label="Expected Answer (Optional)", lines=2)
agent_name_input = gr.Textbox(label="Agent Name", value="Agent-1")
task_type = gr.Dropdown(label="Task Type", choices=["general", "QA", "summarization", "reasoning", "code"], value="general")
evaluate_btn = gr.Button("πŸ” Evaluate", variant="primary")
with gr.Column(scale=1):
scores_output = gr.JSON(label="πŸ“Š Evaluation Scores")
explanation_output = gr.Textbox(label="πŸ“ Explanation", lines=6, interactive=False)
with gr.Row():
spider_chart_output = gr.Plot(label="πŸ•ΈοΈ Spider Chart")
score_bars_output = gr.Plot(label="πŸ“Š Score Breakdown")
evaluate_btn.click(fn=process_single_evaluation, inputs=[prompt_input, response_input, expected_answer, agent_name_input, task_type], outputs=[scores_output, spider_chart_output, score_bars_output, explanation_output])
# Tab 2: Batch Evaluation
with gr.TabItem("πŸ“¦ Batch Evaluation"):
with gr.Row():
with gr.Column(scale=1):
batch_file = gr.File(label="Upload Evaluation File (JSON/JSONL)", file_types=[".json", ".jsonl"])
evaluation_mode = gr.Radio(label="Evaluation Mode", choices=["comprehensive", "fast", "detailed"], value="comprehensive")
batch_evaluate_btn = gr.Button("πŸš€ Run Batch Evaluation", variant="primary")
with gr.Column(scale=2):
batch_report = gr.Textbox(label="πŸ“„ Evaluation Report", lines=10, interactive=False)
with gr.Row():
heatmap_output = gr.Plot(label="πŸ—ΊοΈ Evaluation Heatmap")
distribution_output = gr.Plot(label="πŸ“ˆ Score Distribution")
trends_output = gr.Plot(label="πŸ“Š Performance Trends")
leaderboard_output = gr.Dataframe(label="πŸ† Agent Leaderboard", interactive=False)
batch_evaluate_btn.click(fn=process_batch_evaluation, inputs=[batch_file, evaluation_mode], outputs=[heatmap_output, distribution_output, trends_output, batch_report, leaderboard_output])
# Tab 3: Agent Comparison
with gr.TabItem("βš”οΈ Agent Comparison"):
with gr.Row():
with gr.Column():
agent1_file = gr.File(label="Agent 1 Results (JSON/JSONL)", file_types=[".json", ".jsonl"])
agent2_file = gr.File(label="Agent 2 Results (JSON/JSONL)", file_types=[".json", ".jsonl"])
compare_btn = gr.Button("βš–οΈ Compare Agents", variant="primary")
with gr.Column():
comparison_report = gr.Textbox(label="πŸ“Š Comparison Analysis", lines=8, interactive=False)
with gr.Row():
comparison_chart = gr.Plot(label="πŸ“Š Performance Comparison")
radar_comparison = gr.Plot(label="🎯 Radar Comparison")
performance_delta = gr.Plot(label="πŸ“ˆ Performance Delta Analysis")
compare_btn.click(fn=compare_agents, inputs=[agent1_file, agent2_file], outputs=[comparison_chart, radar_comparison, performance_delta, comparison_report])
# Tab 4: Explainability Dashboard
with gr.TabItem("πŸ” Explainability"):
gr.Markdown("### Deep dive into evaluation decisions. Enter an `eval_id` from a previous run.")
with gr.Row():
eval_id_input = gr.Textbox(label="Evaluation ID", placeholder="Enter evaluation ID to analyze...")
load_eval_btn = gr.Button("Load Evaluation", variant="secondary")
with gr.Row():
eval_details = gr.JSON(label="Evaluation Details")
detailed_breakdown = gr.Markdown(label="Detailed Score Breakdown")
load_eval_btn.click(fn=get_detailed_explanation, inputs=[eval_id_input], outputs=[eval_details, detailed_breakdown])
if __name__ == "__main__":
demo.launch(share=True)