Spaces:
Sleeping
Sleeping
Download app.py from aaditya-raj/e6test: direct link, hf CLI and curl.
- Browser
- Download file 12 kB
-
https://huggingface.co/spaces/aaditya-raj/e6test/resolve/99ea82dd9553f634c8b78472a707c44c53d99b61/app.py
- Command line
-
hf download hf://spaces/aaditya-raj/e6test@99ea82dd9553f634c8b78472a707c44c53d99b61/app.py
-
curl -L -o app.py https://huggingface.co/spaces/aaditya-raj/e6test/resolve/99ea82dd9553f634c8b78472a707c44c53d99b61/app.py
12 kB
| import gradio as gr | |
| import pandas as pd | |
| import numpy as np | |
| import json | |
| import plotly.graph_objects as go | |
| from typing import Dict, List, Tuple, Optional | |
| # Import evaluation modules | |
| from evaluator_module import AetherScoreEvaluator | |
| from visualizer_module import EvaluationVisualizer | |
| from report_generator import ReportGenerator | |
| # --- Global Components & Storage --- | |
| evaluator = AetherScoreEvaluator() | |
| visualizer = EvaluationVisualizer() | |
| report_gen = ReportGenerator() | |
| # In-memory storage for explainability feature | |
| evaluation_storage: Dict[str, Dict] = {} | |
| # CSS for better styling | |
| custom_css = """ | |
| .gradio-container { | |
| font-family: 'Inter', sans-serif; | |
| } | |
| .metric-card { | |
| background: linear-gradient(135deg, #667eea 0%, #764ba2 100%); | |
| padding: 20px; | |
| border-radius: 10px; | |
| color: white; | |
| margin: 10px 0; | |
| } | |
| """ | |
| # Single Evaluation Process | |
| def process_single_evaluation( | |
| prompt: str, | |
| response: str, | |
| expected_answer: Optional[str] = None, | |
| agent_name: str = "Agent-1", | |
| task_type: str = "general" | |
| ) -> Tuple[Dict, go.Figure, go.Figure, str]: | |
| # If Prompt or Response is missing from User | |
| if not prompt or not response: | |
| return {}, go.Figure(), go.Figure(), "Please provide both prompt and response." | |
| # Evaluate the response | |
| eval_result = evaluator.evaluate_single( | |
| prompt=prompt, | |
| response=response, | |
| expected_answer=expected_answer, | |
| task_type=task_type | |
| ) | |
| scores = eval_result.get("scores", {}) | |
| # # Store result for explainability | |
| # eval_id = scores.get("eval_id", "") | |
| # if eval_id: | |
| # evaluation_storage[eval_id] = eval_result | |
| # print(f"Stored evaluation {eval_id}") | |
| # Generate visualizations | |
| spider_chart = visualizer.create_spider_chart(scores, agent_name) | |
| score_bars = visualizer.create_score_bars(scores, agent_name) | |
| # Generating explanation | |
| explanation = evaluator.generate_explanation(scores) | |
| # Format scores for display | |
| scores_display = {k: f"{v:.2f}" for k, v in scores.items() if isinstance(v, float)} | |
| # scores_display["eval_id"] = eval_id | |
| return scores_display, spider_chart, score_bars, explanation | |
| # For Batch Evaluation | |
| def process_batch_evaluation( | |
| file_input, | |
| evaluation_mode: str = "comprehensive" | |
| ) -> Tuple[go.Figure, go.Figure, go.Figure, str, pd.DataFrame]: | |
| # If file input is None | |
| if file_input is None: | |
| return go.Figure(), go.Figure(), go.Figure(), "Please upload a file.", pd.DataFrame() | |
| try: | |
| if file_input.name.endswith('.json'): | |
| with open(file_input.name, 'r') as f: | |
| data = json.load(f) | |
| elif file_input.name.endswith('.jsonl'): | |
| data = [] | |
| with open(file_input.name, 'r') as f: | |
| for line in f: | |
| data.append(json.loads(line)) | |
| else: | |
| raise ValueError("Unsupported file format. Please upload JSON or JSONL.") | |
| # Process batch evaluation | |
| results = evaluator.evaluate_batch(data, mode=evaluation_mode) | |
| # Store all results for explainability | |
| # for res in results: | |
| # eval_id = res.get('scores', {}).get('eval_id') | |
| # if eval_id: | |
| # evaluation_storage[eval_id] = res | |
| # Generating visualizations | |
| heatmap = visualizer.create_evaluation_heatmap(results) | |
| distribution = visualizer.create_score_distribution(results) | |
| trends = visualizer.create_performance_trends(results) | |
| report = report_gen.generate_batch_report(results) | |
| leaderboard = create_leaderboard(results) | |
| return heatmap, distribution, trends, report, leaderboard | |
| except Exception as e: | |
| error_msg = f"Error processing batch file: {str(e)}" | |
| return go.Figure(), go.Figure(), go.Figure(), error_msg, pd.DataFrame() | |
| def create_leaderboard(results: List[Dict]) -> pd.DataFrame: | |
| """Create a leaderboard from evaluation results""" | |
| agent_scores = evaluator.get_agent_scores_from_results(results) | |
| leaderboard_data = [] | |
| for agent, scores in agent_scores.items(): | |
| leaderboard_data.append({ | |
| 'Rank': 0, 'Agent': agent, 'Avg Score': np.mean(scores), | |
| 'Max Score': np.max(scores), 'Min Score': np.min(scores), | |
| 'Std Dev': np.std(scores), 'Evaluations': len(scores) | |
| }) | |
| df = pd.DataFrame(leaderboard_data).sort_values('Avg Score', ascending=False) | |
| df['Rank'] = range(1, len(df) + 1) | |
| for col in ['Avg Score', 'Max Score', 'Min Score', 'Std Dev']: | |
| df[col] = df[col].apply(lambda x: f"{x:.3f}") | |
| return df | |
| def compare_agents( | |
| agent1_file, | |
| agent2_file, | |
| ) -> Tuple[go.Figure, go.Figure, go.Figure, str]: | |
| """Compare two agents' performance""" | |
| if not agent1_file or not agent2_file: | |
| return go.Figure(), go.Figure(), go.Figure(), "Please upload files for both agents." | |
| try: | |
| def load_agent_data(file): | |
| if file.name.endswith('.json'): | |
| with open(file.name, 'r') as f: return json.load(f) | |
| elif file.name.endswith('.jsonl'): | |
| data = []; | |
| with open(file.name, 'r') as f: | |
| for line in f: data.append(json.loads(line)) | |
| return data | |
| raise ValueError("Unsupported file format") | |
| agent1_results = evaluator.evaluate_batch(load_agent_data(agent1_file)) | |
| agent2_results = evaluator.evaluate_batch(load_agent_data(agent2_file)) | |
| comparison_chart = visualizer.create_agent_comparison(agent1_results, agent2_results) | |
| radar_comparison = visualizer.create_radar_comparison(agent1_results, agent2_results) | |
| performance_delta = visualizer.create_performance_delta(agent1_results, agent2_results) | |
| comparison_report = report_gen.generate_comparison_report(agent1_results, agent2_results) | |
| return comparison_chart, radar_comparison, performance_delta, comparison_report | |
| except Exception as e: | |
| error_msg = f"Error comparing agents: {str(e)}" | |
| return go.Figure(), go.Figure(), go.Figure(), error_msg | |
| def get_detailed_explanation(eval_id: str) -> Tuple[Dict, str]: | |
| """Retrieve and format the detailed explanation for an evaluation ID.""" | |
| if not eval_id: | |
| return {}, "Please enter an Evaluation ID." | |
| eval_result = evaluation_storage.get(eval_id) | |
| if not eval_result: | |
| return {}, f"Error: Evaluation ID '{eval_id}' not found in memory." | |
| scores = eval_result.get("scores", {}) | |
| reasons = eval_result.get("reasons", {}) | |
| # Format details for JSON display | |
| details_to_display = { | |
| "evaluation_id": scores.get("eval_id"), | |
| "timestamp": scores.get("timestamp"), | |
| "task_type": scores.get("task_type"), | |
| "scores": {k: v for k, v in scores.items() if k not in ["eval_id", "timestamp", "task_type"]} | |
| } | |
| # Format breakdown for text display | |
| breakdown = [] | |
| breakdown.append(f"### Detailed Breakdown for Evaluation: {eval_id}\n") | |
| for metric, reason in reasons.items(): | |
| metric_name = metric.replace('_', ' ').title() | |
| score = scores.get(metric, 0) | |
| breakdown.append(f"**{metric_name}**: {score:.2f}\n" | |
| f"> *Reasoning*: {reason}\n") | |
| return details_to_display, "\n".join(breakdown) | |
| # --- Gradio Interface --- | |
| with gr.Blocks(title="AetherScore - AI Agent Evaluation Framework", css=custom_css) as demo: | |
| gr.Markdown("# π AetherScore - Advanced AI Agent Evaluation Framework") | |
| with gr.Tabs(): | |
| # Tab 1: Single Evaluation | |
| with gr.TabItem("π― Single Evaluation"): | |
| with gr.Row(): | |
| with gr.Column(scale=1): | |
| prompt_input = gr.Textbox(label="Prompt", lines=3) | |
| response_input = gr.Textbox(label="Agent Response", lines=3) | |
| expected_answer = gr.Textbox(label="Expected Answer (Optional)", lines=2) | |
| agent_name_input = gr.Textbox(label="Agent Name", value="Agent-1") | |
| task_type = gr.Dropdown(label="Task Type", choices=["general", "QA", "summarization", "reasoning", "code"], value="general") | |
| evaluate_btn = gr.Button("π Evaluate", variant="primary") | |
| with gr.Column(scale=1): | |
| scores_output = gr.JSON(label="π Evaluation Scores") | |
| explanation_output = gr.Textbox(label="π Explanation", lines=6, interactive=False) | |
| with gr.Row(): | |
| spider_chart_output = gr.Plot(label="πΈοΈ Spider Chart") | |
| score_bars_output = gr.Plot(label="π Score Breakdown") | |
| evaluate_btn.click(fn=process_single_evaluation, inputs=[prompt_input, response_input, expected_answer, agent_name_input, task_type], outputs=[scores_output, spider_chart_output, score_bars_output, explanation_output]) | |
| # Tab 2: Batch Evaluation | |
| with gr.TabItem("π¦ Batch Evaluation"): | |
| with gr.Row(): | |
| with gr.Column(scale=1): | |
| batch_file = gr.File(label="Upload Evaluation File (JSON/JSONL)", file_types=[".json", ".jsonl"]) | |
| evaluation_mode = gr.Radio(label="Evaluation Mode", choices=["comprehensive", "fast", "detailed"], value="comprehensive") | |
| batch_evaluate_btn = gr.Button("π Run Batch Evaluation", variant="primary") | |
| with gr.Column(scale=2): | |
| batch_report = gr.Textbox(label="π Evaluation Report", lines=10, interactive=False) | |
| with gr.Row(): | |
| heatmap_output = gr.Plot(label="πΊοΈ Evaluation Heatmap") | |
| distribution_output = gr.Plot(label="π Score Distribution") | |
| trends_output = gr.Plot(label="π Performance Trends") | |
| leaderboard_output = gr.Dataframe(label="π Agent Leaderboard", interactive=False) | |
| batch_evaluate_btn.click(fn=process_batch_evaluation, inputs=[batch_file, evaluation_mode], outputs=[heatmap_output, distribution_output, trends_output, batch_report, leaderboard_output]) | |
| # Tab 3: Agent Comparison | |
| with gr.TabItem("βοΈ Agent Comparison"): | |
| with gr.Row(): | |
| with gr.Column(): | |
| agent1_file = gr.File(label="Agent 1 Results (JSON/JSONL)", file_types=[".json", ".jsonl"]) | |
| agent2_file = gr.File(label="Agent 2 Results (JSON/JSONL)", file_types=[".json", ".jsonl"]) | |
| compare_btn = gr.Button("βοΈ Compare Agents", variant="primary") | |
| with gr.Column(): | |
| comparison_report = gr.Textbox(label="π Comparison Analysis", lines=8, interactive=False) | |
| with gr.Row(): | |
| comparison_chart = gr.Plot(label="π Performance Comparison") | |
| radar_comparison = gr.Plot(label="π― Radar Comparison") | |
| performance_delta = gr.Plot(label="π Performance Delta Analysis") | |
| compare_btn.click(fn=compare_agents, inputs=[agent1_file, agent2_file], outputs=[comparison_chart, radar_comparison, performance_delta, comparison_report]) | |
| # Tab 4: Explainability Dashboard | |
| with gr.TabItem("π Explainability"): | |
| gr.Markdown("### Deep dive into evaluation decisions. Enter an `eval_id` from a previous run.") | |
| with gr.Row(): | |
| eval_id_input = gr.Textbox(label="Evaluation ID", placeholder="Enter evaluation ID to analyze...") | |
| load_eval_btn = gr.Button("Load Evaluation", variant="secondary") | |
| with gr.Row(): | |
| eval_details = gr.JSON(label="Evaluation Details") | |
| detailed_breakdown = gr.Markdown(label="Detailed Score Breakdown") | |
| load_eval_btn.click(fn=get_detailed_explanation, inputs=[eval_id_input], outputs=[eval_details, detailed_breakdown]) | |
| if __name__ == "__main__": | |
| demo.launch(share=True) |