aaditya-raj commited on
Commit
0701672
Β·
verified Β·
1 Parent(s): e3a1a9e

Upload app.py

Browse files
Files changed (1) hide show
  1. app.py +274 -0
app.py ADDED
@@ -0,0 +1,274 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import gradio as gr
2
+ import pandas as pd
3
+ import numpy as np
4
+ import json
5
+ import plotly.graph_objects as go
6
+ from typing import Dict, List, Tuple, Optional
7
+
8
+ # Import evaluation modules
9
+ from evaluator_module import AetherScoreEvaluator
10
+ from visualizer_module import EvaluationVisualizer
11
+ from report_generator import ReportGenerator
12
+
13
+ # --- Global Components & Storage ---
14
+ evaluator = AetherScoreEvaluator()
15
+ visualizer = EvaluationVisualizer()
16
+ report_gen = ReportGenerator()
17
+
18
+ # In-memory storage for explainability feature
19
+ evaluation_storage: Dict[str, Dict] = {}
20
+
21
+ # CSS for better styling
22
+ custom_css = """
23
+ .gradio-container {
24
+ font-family: 'Inter', sans-serif;
25
+ }
26
+ .metric-card {
27
+ background: linear-gradient(135deg, #667eea 0%, #764ba2 100%);
28
+ padding: 20px;
29
+ border-radius: 10px;
30
+ color: white;
31
+ margin: 10px 0;
32
+ }
33
+ """
34
+
35
+ # Single Evaluation Process
36
+ def process_single_evaluation(
37
+ prompt: str,
38
+ response: str,
39
+ expected_answer: Optional[str] = None,
40
+ agent_name: str = "Agent-1",
41
+ task_type: str = "general"
42
+ ) -> Tuple[Dict, go.Figure, go.Figure, str]:
43
+
44
+ # If Prompt or Response is missing from User
45
+ if not prompt or not response:
46
+ return {}, go.Figure(), go.Figure(), "Please provide both prompt and response."
47
+
48
+ # Evaluate the response
49
+ eval_result = evaluator.evaluate_single(
50
+ prompt=prompt,
51
+ response=response,
52
+ expected_answer=expected_answer,
53
+ task_type=task_type
54
+ )
55
+ scores = eval_result.get("scores", {})
56
+
57
+ # # Store result for explainability
58
+ # eval_id = scores.get("eval_id", "")
59
+ # if eval_id:
60
+ # evaluation_storage[eval_id] = eval_result
61
+ # print(f"Stored evaluation {eval_id}")
62
+
63
+ # Generate visualizations
64
+ spider_chart = visualizer.create_spider_chart(scores, agent_name)
65
+ score_bars = visualizer.create_score_bars(scores, agent_name)
66
+
67
+ # Generating explanation
68
+ explanation = evaluator.generate_explanation(scores)
69
+
70
+ # Format scores for display
71
+ scores_display = {k: f"{v:.2f}" for k, v in scores.items() if isinstance(v, float)}
72
+ # scores_display["eval_id"] = eval_id
73
+
74
+ return scores_display, spider_chart, score_bars, explanation
75
+
76
+ # For Batch Evaluation
77
+ def process_batch_evaluation(
78
+ file_input,
79
+ evaluation_mode: str = "comprehensive"
80
+ ) -> Tuple[go.Figure, go.Figure, go.Figure, str, pd.DataFrame]:
81
+
82
+ # If file input is None
83
+ if file_input is None:
84
+ return go.Figure(), go.Figure(), go.Figure(), "Please upload a file.", pd.DataFrame()
85
+ try:
86
+ if file_input.name.endswith('.json'):
87
+ with open(file_input.name, 'r') as f:
88
+ data = json.load(f)
89
+ elif file_input.name.endswith('.jsonl'):
90
+ data = []
91
+ with open(file_input.name, 'r') as f:
92
+ for line in f:
93
+ data.append(json.loads(line))
94
+ else:
95
+ raise ValueError("Unsupported file format. Please upload JSON or JSONL.")
96
+
97
+ # Process batch evaluation
98
+ results = evaluator.evaluate_batch(data, mode=evaluation_mode)
99
+
100
+ # Store all results for explainability
101
+ # for res in results:
102
+ # eval_id = res.get('scores', {}).get('eval_id')
103
+ # if eval_id:
104
+ # evaluation_storage[eval_id] = res
105
+
106
+ # Generating visualizations
107
+ heatmap = visualizer.create_evaluation_heatmap(results)
108
+ distribution = visualizer.create_score_distribution(results)
109
+ trends = visualizer.create_performance_trends(results)
110
+ report = report_gen.generate_batch_report(results)
111
+ leaderboard = create_leaderboard(results)
112
+
113
+ return heatmap, distribution, trends, report, leaderboard
114
+
115
+ except Exception as e:
116
+ error_msg = f"Error processing batch file: {str(e)}"
117
+ return go.Figure(), go.Figure(), go.Figure(), error_msg, pd.DataFrame()
118
+
119
+ def create_leaderboard(results: List[Dict]) -> pd.DataFrame:
120
+ """Create a leaderboard from evaluation results"""
121
+ agent_scores = evaluator.get_agent_scores_from_results(results)
122
+
123
+ leaderboard_data = []
124
+ for agent, scores in agent_scores.items():
125
+ leaderboard_data.append({
126
+ 'Rank': 0, 'Agent': agent, 'Avg Score': np.mean(scores),
127
+ 'Max Score': np.max(scores), 'Min Score': np.min(scores),
128
+ 'Std Dev': np.std(scores), 'Evaluations': len(scores)
129
+ })
130
+
131
+ df = pd.DataFrame(leaderboard_data).sort_values('Avg Score', ascending=False)
132
+ df['Rank'] = range(1, len(df) + 1)
133
+
134
+ for col in ['Avg Score', 'Max Score', 'Min Score', 'Std Dev']:
135
+ df[col] = df[col].apply(lambda x: f"{x:.3f}")
136
+
137
+ return df
138
+
139
+ def compare_agents(
140
+ agent1_file,
141
+ agent2_file,
142
+ ) -> Tuple[go.Figure, go.Figure, go.Figure, str]:
143
+ """Compare two agents' performance"""
144
+ if not agent1_file or not agent2_file:
145
+ return go.Figure(), go.Figure(), go.Figure(), "Please upload files for both agents."
146
+
147
+ try:
148
+ def load_agent_data(file):
149
+ if file.name.endswith('.json'):
150
+ with open(file.name, 'r') as f: return json.load(f)
151
+ elif file.name.endswith('.jsonl'):
152
+ data = [];
153
+ with open(file.name, 'r') as f:
154
+ for line in f: data.append(json.loads(line))
155
+ return data
156
+ raise ValueError("Unsupported file format")
157
+
158
+ agent1_results = evaluator.evaluate_batch(load_agent_data(agent1_file))
159
+ agent2_results = evaluator.evaluate_batch(load_agent_data(agent2_file))
160
+
161
+ comparison_chart = visualizer.create_agent_comparison(agent1_results, agent2_results)
162
+ radar_comparison = visualizer.create_radar_comparison(agent1_results, agent2_results)
163
+ performance_delta = visualizer.create_performance_delta(agent1_results, agent2_results)
164
+ comparison_report = report_gen.generate_comparison_report(agent1_results, agent2_results)
165
+
166
+ return comparison_chart, radar_comparison, performance_delta, comparison_report
167
+
168
+ except Exception as e:
169
+ error_msg = f"Error comparing agents: {str(e)}"
170
+ return go.Figure(), go.Figure(), go.Figure(), error_msg
171
+
172
+ def get_detailed_explanation(eval_id: str) -> Tuple[Dict, str]:
173
+ """Retrieve and format the detailed explanation for an evaluation ID."""
174
+ if not eval_id:
175
+ return {}, "Please enter an Evaluation ID."
176
+
177
+ eval_result = evaluation_storage.get(eval_id)
178
+
179
+ if not eval_result:
180
+ return {}, f"Error: Evaluation ID '{eval_id}' not found in memory."
181
+
182
+ scores = eval_result.get("scores", {})
183
+ reasons = eval_result.get("reasons", {})
184
+
185
+ # Format details for JSON display
186
+ details_to_display = {
187
+ "evaluation_id": scores.get("eval_id"),
188
+ "timestamp": scores.get("timestamp"),
189
+ "task_type": scores.get("task_type"),
190
+ "scores": {k: v for k, v in scores.items() if k not in ["eval_id", "timestamp", "task_type"]}
191
+ }
192
+
193
+ # Format breakdown for text display
194
+ breakdown = []
195
+ breakdown.append(f"### Detailed Breakdown for Evaluation: {eval_id}\n")
196
+ for metric, reason in reasons.items():
197
+ metric_name = metric.replace('_', ' ').title()
198
+ score = scores.get(metric, 0)
199
+ breakdown.append(f"**{metric_name}**: {score:.2f}\n"
200
+ f"> *Reasoning*: {reason}\n")
201
+
202
+ return details_to_display, "\n".join(breakdown)
203
+
204
+ # --- Gradio Interface ---
205
+ with gr.Blocks(title="AetherScore - AI Agent Evaluation Framework", css=custom_css) as demo:
206
+ gr.Markdown("# 🌟 AetherScore - Advanced AI Agent Evaluation Framework")
207
+
208
+ with gr.Tabs():
209
+ # Tab 1: Single Evaluation
210
+ with gr.TabItem("🎯 Single Evaluation"):
211
+ with gr.Row():
212
+ with gr.Column(scale=1):
213
+ prompt_input = gr.Textbox(label="Prompt", lines=3)
214
+ response_input = gr.Textbox(label="Agent Response", lines=3)
215
+ expected_answer = gr.Textbox(label="Expected Answer (Optional)", lines=2)
216
+ agent_name_input = gr.Textbox(label="Agent Name", value="Agent-1")
217
+ task_type = gr.Dropdown(label="Task Type", choices=["general", "QA", "summarization", "reasoning", "code"], value="general")
218
+ evaluate_btn = gr.Button("πŸ” Evaluate", variant="primary")
219
+ with gr.Column(scale=1):
220
+ scores_output = gr.JSON(label="πŸ“Š Evaluation Scores")
221
+ explanation_output = gr.Textbox(label="πŸ“ Explanation", lines=6, interactive=False)
222
+ with gr.Row():
223
+ spider_chart_output = gr.Plot(label="πŸ•ΈοΈ Spider Chart")
224
+ score_bars_output = gr.Plot(label="πŸ“Š Score Breakdown")
225
+
226
+ evaluate_btn.click(fn=process_single_evaluation, inputs=[prompt_input, response_input, expected_answer, agent_name_input, task_type], outputs=[scores_output, spider_chart_output, score_bars_output, explanation_output])
227
+
228
+ # Tab 2: Batch Evaluation
229
+ with gr.TabItem("πŸ“¦ Batch Evaluation"):
230
+ with gr.Row():
231
+ with gr.Column(scale=1):
232
+ batch_file = gr.File(label="Upload Evaluation File (JSON/JSONL)", file_types=[".json", ".jsonl"])
233
+ evaluation_mode = gr.Radio(label="Evaluation Mode", choices=["comprehensive", "fast", "detailed"], value="comprehensive")
234
+ batch_evaluate_btn = gr.Button("πŸš€ Run Batch Evaluation", variant="primary")
235
+ with gr.Column(scale=2):
236
+ batch_report = gr.Textbox(label="πŸ“„ Evaluation Report", lines=10, interactive=False)
237
+ with gr.Row():
238
+ heatmap_output = gr.Plot(label="πŸ—ΊοΈ Evaluation Heatmap")
239
+ distribution_output = gr.Plot(label="πŸ“ˆ Score Distribution")
240
+ trends_output = gr.Plot(label="πŸ“Š Performance Trends")
241
+ leaderboard_output = gr.Dataframe(label="πŸ† Agent Leaderboard", interactive=False)
242
+
243
+ batch_evaluate_btn.click(fn=process_batch_evaluation, inputs=[batch_file, evaluation_mode], outputs=[heatmap_output, distribution_output, trends_output, batch_report, leaderboard_output])
244
+
245
+ # Tab 3: Agent Comparison
246
+ with gr.TabItem("βš”οΈ Agent Comparison"):
247
+ with gr.Row():
248
+ with gr.Column():
249
+ agent1_file = gr.File(label="Agent 1 Results (JSON/JSONL)", file_types=[".json", ".jsonl"])
250
+ agent2_file = gr.File(label="Agent 2 Results (JSON/JSONL)", file_types=[".json", ".jsonl"])
251
+ compare_btn = gr.Button("βš–οΈ Compare Agents", variant="primary")
252
+ with gr.Column():
253
+ comparison_report = gr.Textbox(label="πŸ“Š Comparison Analysis", lines=8, interactive=False)
254
+ with gr.Row():
255
+ comparison_chart = gr.Plot(label="πŸ“Š Performance Comparison")
256
+ radar_comparison = gr.Plot(label="🎯 Radar Comparison")
257
+ performance_delta = gr.Plot(label="πŸ“ˆ Performance Delta Analysis")
258
+
259
+ compare_btn.click(fn=compare_agents, inputs=[agent1_file, agent2_file], outputs=[comparison_chart, radar_comparison, performance_delta, comparison_report])
260
+
261
+ # Tab 4: Explainability Dashboard
262
+ with gr.TabItem("πŸ” Explainability"):
263
+ gr.Markdown("### Deep dive into evaluation decisions. Enter an `eval_id` from a previous run.")
264
+ with gr.Row():
265
+ eval_id_input = gr.Textbox(label="Evaluation ID", placeholder="Enter evaluation ID to analyze...")
266
+ load_eval_btn = gr.Button("Load Evaluation", variant="secondary")
267
+ with gr.Row():
268
+ eval_details = gr.JSON(label="Evaluation Details")
269
+ detailed_breakdown = gr.Markdown(label="Detailed Score Breakdown")
270
+
271
+ load_eval_btn.click(fn=get_detailed_explanation, inputs=[eval_id_input], outputs=[eval_details, detailed_breakdown])
272
+
273
+ if __name__ == "__main__":
274
+ demo.launch(share=True)