aaditya-raj commited on
Commit
d29d0c6
·
verified ·
1 Parent(s): 92eef9e

Update evaluator_module.py

Browse files
Files changed (1) hide show
  1. evaluator_module.py +348 -197
evaluator_module.py CHANGED
@@ -14,6 +14,7 @@ import hashlib
14
  from datetime import datetime
15
  import concurrent.futures
16
  import random
 
17
 
18
  class AetherScoreEvaluator:
19
  def __init__(self):
@@ -25,245 +26,395 @@ class AetherScoreEvaluator:
25
  spacy.cli.download("en_core_web_sm")
26
  self.nlp = spacy.load("en_core_web_sm")
27
 
28
- # LLM Judge Model
29
- self.judge_model = pipeline(
30
- "text2text-generation",
31
- model="google/flan-t5-base",
32
- device=-1 # 0 for GPU
33
- )
34
-
35
- # Sentence Transformer for Sentence ----> Embedding
36
- self.sentence_model = SentenceTransformer('all-MiniLM-L6-v2')
37
-
38
- # for hallucination
39
- self.rouge = evaluate.load("rouge")
40
- self.sacrebleu = evaluate.load("sacrebleu")
41
- self.nli_tokenizer = AutoTokenizer.from_pretrained("prajjwal1/bert-mini-mnli")
42
- self.nli_model = AutoModelForSequenceClassification.from_pretrained("prajjwal1/bert-mini-mnli")
43
-
44
- # Scoring weights # Domain Specific weights can be added for better results
45
- self.weights = {'instruction_following': 0.25, 'hallucination_score': 0.20,
46
- 'assumption_control': 0.20, 'coherence': 0.20, 'accuracy': 0.15}
47
 
48
  # In-memory cache
49
  self.cache = {}
50
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
51
  def _evaluate_with_llm_judge(self, prompt: str, response: str) -> dict:
52
  """
53
- Hallucination detection using:
54
- - NLI (entailment, neutral, contradiction)
55
- - Embedding similarity
56
- - ROUGE-L
57
- - SacreBLEU
58
- Assumption control derived from NLI neutrality.
59
  """
60
- # Step 1: Embedding similarity
61
- emb_sim = self._semantic_similarity(prompt, response)
62
-
63
- # Step 2: NLI inference
64
- inputs = self.nli_tokenizer.encode_plus(prompt, response, return_tensors="pt", truncation=True)
65
- with torch.no_grad():
66
- logits = self.nli_model(**inputs).logits
67
- probs = torch.softmax(logits, dim=-1).cpu().numpy()[0]
68
- entailment, neutral, contradiction = probs[2], probs[1], probs[0]
69
-
70
- # Step 3: ROUGE-L
71
- rouge_l = self.rouge.compute(predictions=[response], references=[prompt])["rougeL"]
72
-
73
- # Step 4: SacreBLEU (normalized 0–1)
74
- sacrebleu = self.sacrebleu.compute(predictions=[response], references=[[prompt]])["score"] / 100.0
75
-
76
- # Step 5: Weighted hallucination score
77
- weights = {"entailment": 0.4, "embedding": 0.2, "rouge": 0.2, "sacrebleu": 0.2}
78
- halluc_score = 1 - (
79
- weights["entailment"] * entailment +
80
- weights["embedding"] * emb_sim +
81
- weights["rouge"] * rouge_l +
82
- weights["sacrebleu"] * sacrebleu
83
- )
84
-
85
- # Step 6: Assumption control from neutrality
86
- assumption_score = 1 - neutral
87
-
88
- # Step 7: Explanations
89
- halluc_expl = (
90
- f"Entailment={entailment:.2f}, Embedding={emb_sim:.2f}, "
91
- f"ROUGE-L={rouge_l:.2f}, SacreBLEU={sacrebleu:.2f}, Neutral={neutral:.2f}"
92
- )
93
- assumption_expl = (
94
- f"Assumption control is derived from NLI neutrality={neutral:.2f}. "
95
- "Lower neutrality → stronger confidence."
96
- )
97
-
98
- return {
99
- "hallucination_score": (float(halluc_score), halluc_expl),
100
- "assumption_control": (float(assumption_score), assumption_expl),
101
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
102
 
103
- # Single Evaluation # Inputs-->> Prompt, Agent Response, Expected Answer(Optional), Agent Name and Task type( General, QA, Summarizaton)etc
104
  def evaluate_single(self, prompt: str, response: str, expected_answer: Optional[str] = None, task_type: str = "general") -> Dict:
105
-
106
- # Generating Eval ID
107
- eval_id = self._generate_eval_id(prompt, response)
108
-
109
- # If already stored in cache direclty we can return from there.
110
- # if eval_id in self.cache:
111
- # return self.cache[eval_id]
 
112
 
113
- scores, reasons = {}, {}
 
 
 
114
 
115
- # Taking Back scores and reasons of hallucination and assumption control from LLM Judge
116
- llm_judge_results = self._evaluate_with_llm_judge(prompt, response)
117
- scores['hallucination_score'], reasons['hallucination_score'] = llm_judge_results['hallucination_score']
118
- scores['assumption_control'], reasons['assumption_control'] = llm_judge_results['assumption_control']
119
 
120
- # Evaluating Instruction Following, Coherence and Accuracy
121
- scores['instruction_following'], reasons['instruction_following'] = self._evaluate_instruction_following(prompt, response)
122
- scores['coherence'], reasons['coherence'] = self._evaluate_coherence(response)
123
- scores['accuracy'], reasons['accuracy'] = self._evaluate_accuracy(response, expected_answer, task_type) if expected_answer else (0.5, "No expected answer provided.")
 
 
 
 
124
 
125
- # Calculating Overall Score
126
- scores['overall_score'] = self._calculate_overall_score(scores)
127
- reasons['overall_score'] = f" Weighted Average Score based on component scores."
128
 
129
- # Updating Eval ID, Timestamp and Task Type in Scores
130
- scores.update({'eval_id': eval_id, 'timestamp': datetime.now().isoformat(), 'task_type': task_type})
 
 
 
 
131
 
132
- # Updating scores(Eval ID, timestamp and task_type) and reasons(all scores) in result
133
- result = {"scores": scores, "reasons": reasons}
134
 
135
- #Storing results with corresponding Eval ID in cache
136
- # self.cache[eval_id] = result
137
-
138
- return result
 
 
139
 
140
- # Batch Evaluation # Input of JSON/CSV file
141
  def evaluate_batch(self, data: List[Dict], mode: str = "comprehensive") -> List[Dict]:
142
- """Process a batch of evaluations in parallel."""
143
-
 
 
144
  results = []
 
145
 
146
- # Get Item function
147
  def process_item(item):
148
- # Calling our Evalution function for Single prompt response pair
149
- eval_result = self.evaluate_single(
150
- prompt=item.get('prompt', ''),
151
- response=item.get('response', ''),
152
- expected_answer=item.get('expected_answer',''),
153
- task_type=item.get('task_type', 'general')
154
- )
155
- # Combining with original metadata
156
- eval_result.update({
157
- 'task_id': item.get('task_id', eval_result['scores']['eval_id']),
158
- 'agent_name': item.get('agent_name', 'Unknown'),
159
- })
160
- return eval_result
161
 
162
- with concurrent.futures.ThreadPoolExecutor() as executor:
163
- future_to_item = {executor.submit(process_item, item): item for item in data}
164
- for future in concurrent.futures.as_completed(future_to_item):
 
 
 
 
 
165
  try:
166
- results.append(future.result())
 
 
 
 
 
 
 
 
 
167
  except Exception as exc:
168
- print(f'An item generated an exception: {exc}')
 
 
 
 
 
 
 
 
169
 
170
  return results
171
-
172
- # Instruction Following Evaluation (Prompt, Response)
173
  def _evaluate_instruction_following(self, prompt: str, response: str) -> Tuple[float, str]:
174
- score, checks, passed = 1.0, 0, 0
175
-
176
- # Check for negative constraints
177
- negations = re.findall(r"(don't|do not|avoid|without) ([\w\s,]+)", prompt.lower())
178
- for _, constraint_phrase in negations:
179
- checks += 1
180
- words_to_avoid = [w.strip() for w in constraint_phrase.split(',')]
181
- if not any(word in response.lower() for word in words_to_avoid if len(word) > 2):
182
- passed += 1
183
-
184
- # Fallback to semantic similarity if no specific instructions found
185
- if checks == 0:
186
- sim = self._semantic_similarity(prompt, response)
187
- return sim, f"No specific constraints found. Score based on semantic similarity ({sim:.2f}) to prompt."
188
-
189
- # Final Score calculation
190
- score = passed / checks if checks > 0 else 1.0
191
- reason = f"{passed}/{checks} specific constraints were followed."
192
-
193
- return score, reason
 
 
 
 
194
 
195
- # Evaluating Coherence (response)
196
  def _evaluate_coherence(self, response: str) -> Tuple[float, str]:
 
 
 
 
197
 
198
- # Extracting Sentences from Response
199
- doc = self.nlp(response)
200
- sentences = [sent.text for sent in doc.sents]
201
-
202
- # If only one Sentence then Coherence is Neutral
203
- if len(sentences) < 2:
204
- return 0.7, "Coherence is neutral for single-sentence responses."
205
 
206
- # Fetching Embeddings from our Sentence Model
207
- embeddings = self.sentence_model.encode(sentences)
208
- sims = [cosine_similarity([embeddings[i]], [embeddings[i+1]])[0][0] for i in range(len(sentences)-1)]
209
-
210
- score = np.mean(sims)
211
-
212
- reason = f"Average sentence-to-sentence similarity score is {score:.2f} across {len(sentences)} sentences."
213
- return score, reason
 
 
 
 
214
 
215
- # Evaluating Accuracy (Response, Expected, Task_type)
216
  def _evaluate_accuracy(self, response: str, expected: str, task_type: str) -> Tuple[float, str]:
217
- sim = self._semantic_similarity(response, expected)
218
- reason = f"Semantic similarity between response and expected answer is {sim:.2f}."
219
- if sim > 0.95:
220
- reason += " (High match)"
221
- elif sim < 0.5:
222
- reason += " (Low match)"
223
- return sim, reason
224
-
225
- # Overall Score
 
 
 
226
  def _calculate_overall_score(self, scores: Dict) -> float:
227
- total, weight_sum = 0.0, 0.0
228
- for metric, weight in self.weights.items():
229
- if metric in scores:
230
- total += scores[metric] * weight
231
- weight_sum += weight
232
- return total / weight_sum #if weight_sum > 0 else 0.5
233
-
234
- # Explanation Generator, work in progress
 
 
 
235
  def generate_explanation(self, scores: Dict) -> str:
236
- explanation = []
237
- overall = scores.get('overall_score', 0)
238
- explanation.append(f"Overall Score: {overall:.2f}/1.00 - Reflects a weighted average of all dimensions.")
239
-
240
- if scores.get('instruction_following', 0) < 0.6:
241
- explanation.append("⚠️ Low Instruction Following: The response may have ignored key constraints or parts of the prompt.")
242
- if scores.get('hallucination_score', 0) < 0.6:
243
- explanation.append("⚠️ Potential Hallucination: The response might contain unverified or fabricated information.")
244
- if scores.get('accuracy', 0) < 0.6 and scores.get('accuracy', 0.5) != 0.5:
245
- explanation.append("⚠️ Low Accuracy: The response significantly differs from the provided expected answer.")
246
-
247
- if not explanation[1:]:
248
- explanation.append("✅ Great Performance: The agent performed well across the primary evaluation dimensions.")
 
 
249
 
250
- return "\n".join(explanation)
 
 
251
 
252
- # Agent Scores
253
  def get_agent_scores_from_results(self, results: List[Dict]) -> Dict[str, List[float]]:
 
254
  agent_scores = defaultdict(list)
255
  for result in results:
256
- agent_name = result.get('agent_name', 'Unknown')
257
- overall_score = result.get('scores', {}).get('overall_score', 0)
258
- agent_scores[agent_name].append(overall_score)
 
 
 
 
 
259
  return agent_scores
260
 
261
- # Some Helper Functions
262
  def _generate_eval_id(self, prompt: str, response: str) -> str:
263
- return hashlib.md5(f"{prompt}{response}".encode()).hexdigest()[:12]
 
 
 
 
264
 
265
  def _semantic_similarity(self, text1: str, text2: str) -> float:
266
- if not text1 or not text2: return 0.0
267
- emb1 = self.sentence_model.encode([text1])
268
- emb2 = self.sentence_model.encode([text2])
269
- return cosine_similarity(emb1, emb2)[0][0]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
14
  from datetime import datetime
15
  import concurrent.futures
16
  import random
17
+ import gc
18
 
19
  class AetherScoreEvaluator:
20
  def __init__(self):
 
26
  spacy.cli.download("en_core_web_sm")
27
  self.nlp = spacy.load("en_core_web_sm")
28
 
29
+ # Initialize models with error handling
30
+ self._initialize_models()
31
+
32
+ # Scoring weights
33
+ self.weights = {
34
+ 'instruction_following': 0.25,
35
+ 'hallucination_score': 0.20,
36
+ 'assumption_control': 0.20,
37
+ 'coherence': 0.20,
38
+ 'accuracy': 0.15
39
+ }
 
 
 
 
 
 
 
 
40
 
41
  # In-memory cache
42
  self.cache = {}
43
 
44
+ def _initialize_models(self):
45
+ """Initialize all models with proper error handling"""
46
+ try:
47
+ # LLM Judge Model
48
+ self.judge_model = pipeline(
49
+ "text2text-generation",
50
+ model="google/flan-t5-base",
51
+ device=-1 # CPU only for stability
52
+ )
53
+
54
+ # Sentence Transformer
55
+ self.sentence_model = SentenceTransformer('all-MiniLM-L6-v2')
56
+
57
+ # Evaluation metrics
58
+ self.rouge = evaluate.load("rouge")
59
+ self.sacrebleu = evaluate.load("sacrebleu")
60
+
61
+ # NLI models
62
+ self.nli_tokenizer = AutoTokenizer.from_pretrained("prajjwal1/bert-mini-mnli")
63
+ self.nli_model = AutoModelForSequenceClassification.from_pretrained("prajjwal1/bert-mini-mnli")
64
+
65
+ print("All models initialized successfully")
66
+
67
+ except Exception as e:
68
+ print(f"Error initializing models: {e}")
69
+ # Fallback to basic functionality
70
+ self._use_fallback_models()
71
+
72
+ def _use_fallback_models(self):
73
+ """Fallback to basic evaluation if model loading fails"""
74
+ print("Using fallback evaluation methods")
75
+ self.judge_model = None
76
+ self.sentence_model = None
77
+ self.rouge = None
78
+ self.sacrebleu = None
79
+ self.nli_tokenizer = None
80
+ self.nli_model = None
81
+
82
+ def _cleanup_models(self):
83
+ """Clean up model memory"""
84
+ if hasattr(self, 'nli_model') and self.nli_model is not None:
85
+ del self.nli_model
86
+ if hasattr(self, 'judge_model') and self.judge_model is not None:
87
+ del self.judge_model
88
+ torch.cuda.empty_cache() if torch.cuda.is_available() else None
89
+ gc.collect()
90
+
91
  def _evaluate_with_llm_judge(self, prompt: str, response: str) -> dict:
92
  """
93
+ Hallucination detection with robust error handling
 
 
 
 
 
94
  """
95
+ try:
96
+ # Step 1: Embedding similarity (with fallback)
97
+ if self.sentence_model is not None:
98
+ emb_sim = self._semantic_similarity(prompt, response)
99
+ else:
100
+ emb_sim = 0.5 # neutral fallback
101
+
102
+ # Step 2: NLI check (with error handling)
103
+ if self.nli_tokenizer is not None and self.nli_model is not None:
104
+ try:
105
+ inputs = self.nli_tokenizer.encode_plus(
106
+ prompt, response,
107
+ return_tensors="pt",
108
+ truncation=True,
109
+ max_length=512 # Limit token length
110
+ )
111
+ with torch.no_grad():
112
+ logits = self.nli_model(**inputs).logits
113
+ probs = torch.softmax(logits, dim=-1).cpu().numpy()[0]
114
+ entailment, neutral, contradiction = probs[2], probs[1], probs[0]
115
+ except Exception as nli_error:
116
+ print(f"NLI evaluation failed: {nli_error}")
117
+ entailment, neutral, contradiction = 0.33, 0.33, 0.34
118
+ else:
119
+ entailment, neutral, contradiction = 0.33, 0.33, 0.34
120
+
121
+ # Step 3: ROUGE-L (with error handling)
122
+ if self.rouge is not None:
123
+ try:
124
+ rouge_l = self.rouge.compute(predictions=[response], references=[prompt])["rougeL"]
125
+ except Exception as rouge_error:
126
+ print(f"ROUGE evaluation failed: {rouge_error}")
127
+ rouge_l = 0.5
128
+ else:
129
+ rouge_l = 0.5
130
+
131
+ # Step 4: SacreBLEU (with error handling)
132
+ if self.sacrebleu is not None:
133
+ try:
134
+ sacrebleu = self.sacrebleu.compute(predictions=[response], references=[[prompt]])["score"] / 100.0
135
+ except Exception as bleu_error:
136
+ print(f"BLEU evaluation failed: {bleu_error}")
137
+ sacrebleu = 0.5
138
+ else:
139
+ sacrebleu = 0.5
140
+
141
+ # Step 5: Weighted hallucination score
142
+ weights = {"entailment": 0.4, "embedding": 0.2, "rouge": 0.2, "sacrebleu": 0.2}
143
+ halluc_score = 1 - (
144
+ weights["entailment"] * entailment +
145
+ weights["embedding"] * emb_sim +
146
+ weights["rouge"] * rouge_l +
147
+ weights["sacrebleu"] * sacrebleu
148
+ )
149
+
150
+ # Step 6: Assumption control from neutrality
151
+ assumption_score = 1 - neutral
152
+
153
+ # Ensure scores are in valid range
154
+ halluc_score = max(0.0, min(1.0, float(halluc_score)))
155
+ assumption_score = max(0.0, min(1.0, float(assumption_score)))
156
+
157
+ # Step 7: Explanations
158
+ halluc_expl = (
159
+ f"Entailment={entailment:.2f}, Embedding={emb_sim:.2f}, "
160
+ f"ROUGE-L={rouge_l:.2f}, SacreBLEU={sacrebleu:.2f}, Neutral={neutral:.2f}"
161
+ )
162
+ assumption_expl = (
163
+ f"Assumption control derived from NLI neutrality={neutral:.2f}. "
164
+ "Lower neutrality → stronger confidence."
165
+ )
166
+
167
+ return {
168
+ "hallucination_score": (halluc_score, halluc_expl),
169
+ "assumption_control": (assumption_score, assumption_expl),
170
+ }
171
+
172
+ except Exception as e:
173
+ print(f"Evaluation error: {e}")
174
+ # Return fallback scores
175
+ return {
176
+ "hallucination_score": (0.5, f"Evaluation failed: {str(e)}"),
177
+ "assumption_control": (0.5, f"Evaluation failed: {str(e)}"),
178
+ }
179
 
 
180
  def evaluate_single(self, prompt: str, response: str, expected_answer: Optional[str] = None, task_type: str = "general") -> Dict:
181
+ """Single evaluation with enhanced error handling"""
182
+ try:
183
+ # Input validation
184
+ if not prompt or not response:
185
+ return {
186
+ "scores": {"overall_score": 0.0},
187
+ "reasons": {"error": "Empty prompt or response"}
188
+ }
189
 
190
+ # Generating Eval ID
191
+ eval_id = self._generate_eval_id(prompt, response)
192
+
193
+ scores, reasons = {}, {}
194
 
195
+ # LLM Judge evaluation
196
+ llm_judge_results = self._evaluate_with_llm_judge(prompt, response)
197
+ scores['hallucination_score'], reasons['hallucination_score'] = llm_judge_results['hallucination_score']
198
+ scores['assumption_control'], reasons['assumption_control'] = llm_judge_results['assumption_control']
199
 
200
+ # Other evaluations
201
+ scores['instruction_following'], reasons['instruction_following'] = self._evaluate_instruction_following(prompt, response)
202
+ scores['coherence'], reasons['coherence'] = self._evaluate_coherence(response)
203
+
204
+ if expected_answer:
205
+ scores['accuracy'], reasons['accuracy'] = self._evaluate_accuracy(response, expected_answer, task_type)
206
+ else:
207
+ scores['accuracy'], reasons['accuracy'] = (0.5, "No expected answer provided.")
208
 
209
+ # Calculate overall score
210
+ scores['overall_score'] = self._calculate_overall_score(scores)
211
+ reasons['overall_score'] = "Weighted average of component scores."
212
 
213
+ # Add metadata
214
+ scores.update({
215
+ 'eval_id': eval_id,
216
+ 'timestamp': datetime.now().isoformat(),
217
+ 'task_type': task_type
218
+ })
219
 
220
+ return {"scores": scores, "reasons": reasons}
 
221
 
222
+ except Exception as e:
223
+ print(f"Single evaluation error: {e}")
224
+ return {
225
+ "scores": {"overall_score": 0.0, "eval_id": "error"},
226
+ "reasons": {"error": str(e)}
227
+ }
228
 
 
229
  def evaluate_batch(self, data: List[Dict], mode: str = "comprehensive") -> List[Dict]:
230
+ """Process batch with improved error handling and cleanup"""
231
+ if not data:
232
+ return []
233
+
234
  results = []
235
+ failed_count = 0
236
 
 
237
  def process_item(item):
238
+ try:
239
+ return self.evaluate_single(
240
+ prompt=item.get('prompt', ''),
241
+ response=item.get('response', ''),
242
+ expected_answer=item.get('expected_answer', ''),
243
+ task_type=item.get('task_type', 'general')
244
+ )
245
+ except Exception as e:
246
+ print(f"Item processing failed: {e}")
247
+ return None
248
+
249
+ # Use smaller thread pool and add timeout
250
+ max_workers = min(4, len(data)) # Limit concurrent threads
251
 
252
+ with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor:
253
+ # Submit all tasks with timeout
254
+ future_to_item = {
255
+ executor.submit(process_item, item): (i, item)
256
+ for i, item in enumerate(data)
257
+ }
258
+
259
+ for future in concurrent.futures.as_completed(future_to_item, timeout=300): # 5 minute timeout
260
  try:
261
+ result = future.result(timeout=30) # 30 second per item timeout
262
+ if result:
263
+ idx, item = future_to_item[future]
264
+ result.update({
265
+ 'task_id': item.get('task_id', result['scores'].get('eval_id', f'task_{idx}')),
266
+ 'agent_name': item.get('agent_name', 'Unknown'),
267
+ })
268
+ results.append(result)
269
+ else:
270
+ failed_count += 1
271
  except Exception as exc:
272
+ failed_count += 1
273
+ print(f'Item generated exception: {exc}')
274
+
275
+ if failed_count > 0:
276
+ print(f"Warning: {failed_count} items failed to process")
277
+
278
+ # Cleanup after batch processing
279
+ if len(data) > 10: # Only cleanup for larger batches
280
+ gc.collect()
281
 
282
  return results
283
+
 
284
  def _evaluate_instruction_following(self, prompt: str, response: str) -> Tuple[float, str]:
285
+ """Evaluate instruction following with better error handling"""
286
+ try:
287
+ score, checks, passed = 1.0, 0, 0
288
+
289
+ # Check for negative constraints
290
+ negations = re.findall(r"(don't|do not|avoid|without) ([\w\s,]+)", prompt.lower())
291
+ for _, constraint_phrase in negations:
292
+ checks += 1
293
+ words_to_avoid = [w.strip() for w in constraint_phrase.split(',')]
294
+ if not any(word in response.lower() for word in words_to_avoid if len(word) > 2):
295
+ passed += 1
296
+
297
+ # Fallback to semantic similarity if no specific instructions found
298
+ if checks == 0:
299
+ sim = self._semantic_similarity(prompt, response)
300
+ return sim, f"No specific constraints found. Score based on semantic similarity ({sim:.2f}) to prompt."
301
+
302
+ score = passed / checks if checks > 0 else 1.0
303
+ reason = f"{passed}/{checks} specific constraints were followed."
304
+
305
+ return score, reason
306
+
307
+ except Exception as e:
308
+ return 0.5, f"Instruction evaluation failed: {str(e)}"
309
 
 
310
  def _evaluate_coherence(self, response: str) -> Tuple[float, str]:
311
+ """Evaluate coherence with error handling"""
312
+ try:
313
+ if not response.strip():
314
+ return 0.1, "Empty response"
315
 
316
+ doc = self.nlp(response)
317
+ sentences = [sent.text for sent in doc.sents if sent.text.strip()]
318
+
319
+ if len(sentences) < 2:
320
+ return 0.7, "Coherence is neutral for single-sentence responses."
 
 
321
 
322
+ if self.sentence_model is not None:
323
+ embeddings = self.sentence_model.encode(sentences)
324
+ sims = [cosine_similarity([embeddings[i]], [embeddings[i+1]])[0][0] for i in range(len(sentences)-1)]
325
+ score = np.mean(sims)
326
+ else:
327
+ score = 0.7 # fallback
328
+
329
+ reason = f"Average sentence-to-sentence similarity score is {score:.2f} across {len(sentences)} sentences."
330
+ return float(score), reason
331
+
332
+ except Exception as e:
333
+ return 0.5, f"Coherence evaluation failed: {str(e)}"
334
 
 
335
  def _evaluate_accuracy(self, response: str, expected: str, task_type: str) -> Tuple[float, str]:
336
+ """Evaluate accuracy with error handling"""
337
+ try:
338
+ sim = self._semantic_similarity(response, expected)
339
+ reason = f"Semantic similarity between response and expected answer is {sim:.2f}."
340
+ if sim > 0.95:
341
+ reason += " (High match)"
342
+ elif sim < 0.5:
343
+ reason += " (Low match)"
344
+ return sim, reason
345
+ except Exception as e:
346
+ return 0.5, f"Accuracy evaluation failed: {str(e)}"
347
+
348
  def _calculate_overall_score(self, scores: Dict) -> float:
349
+ """Calculate overall score with error handling"""
350
+ try:
351
+ total, weight_sum = 0.0, 0.0
352
+ for metric, weight in self.weights.items():
353
+ if metric in scores and isinstance(scores[metric], (int, float)):
354
+ total += float(scores[metric]) * weight
355
+ weight_sum += weight
356
+ return total / weight_sum if weight_sum > 0 else 0.5
357
+ except Exception:
358
+ return 0.5
359
+
360
  def generate_explanation(self, scores: Dict) -> str:
361
+ """Generate explanation with error handling"""
362
+ try:
363
+ explanation = []
364
+ overall = scores.get('overall_score', 0)
365
+ explanation.append(f"Overall Score: {overall:.2f}/1.00 - Reflects a weighted average of all dimensions.")
366
+
367
+ if scores.get('instruction_following', 0) < 0.6:
368
+ explanation.append("Low Instruction Following: The response may have ignored key constraints or parts of the prompt.")
369
+ if scores.get('hallucination_score', 0) < 0.6:
370
+ explanation.append("Potential Hallucination: The response might contain unverified or fabricated information.")
371
+ if scores.get('accuracy', 0) < 0.6 and scores.get('accuracy', 0.5) != 0.5:
372
+ explanation.append("Low Accuracy: The response significantly differs from the provided expected answer.")
373
+
374
+ if len(explanation) == 1:
375
+ explanation.append("Great Performance: The agent performed well across the primary evaluation dimensions.")
376
 
377
+ return "\n".join(explanation)
378
+ except Exception as e:
379
+ return f"Explanation generation failed: {str(e)}"
380
 
 
381
  def get_agent_scores_from_results(self, results: List[Dict]) -> Dict[str, List[float]]:
382
+ """Get agent scores with error handling"""
383
  agent_scores = defaultdict(list)
384
  for result in results:
385
+ try:
386
+ agent_name = result.get('agent_name', 'Unknown')
387
+ overall_score = result.get('scores', {}).get('overall_score', 0)
388
+ if isinstance(overall_score, (int, float)) and not np.isnan(overall_score):
389
+ agent_scores[agent_name].append(float(overall_score))
390
+ except Exception as e:
391
+ print(f"Error processing result: {e}")
392
+ continue
393
  return agent_scores
394
 
 
395
  def _generate_eval_id(self, prompt: str, response: str) -> str:
396
+ """Generate evaluation ID"""
397
+ try:
398
+ return hashlib.md5(f"{prompt}{response}".encode()).hexdigest()[:12]
399
+ except Exception:
400
+ return hashlib.md5(f"fallback{datetime.now()}".encode()).hexdigest()[:12]
401
 
402
  def _semantic_similarity(self, text1: str, text2: str) -> float:
403
+ """Calculate semantic similarity with error handling"""
404
+ try:
405
+ if not text1 or not text2 or self.sentence_model is None:
406
+ return 0.0
407
+ emb1 = self.sentence_model.encode([text1])
408
+ emb2 = self.sentence_model.encode([text2])
409
+ sim = cosine_similarity(emb1, emb2)[0][0]
410
+ return float(sim) if not np.isnan(sim) else 0.0
411
+ except Exception as e:
412
+ print(f"Similarity calculation failed: {e}")
413
+ return 0.0
414
+
415
+ def __del__(self):
416
+ """Cleanup when object is destroyed"""
417
+ try:
418
+ self._cleanup_models()
419
+ except Exception:
420
+ pass