RhutuTuvoc commited on
Commit
b9d3da5
·
1 Parent(s): b847937

Strengthen reward penalties for undesirable agent behavior

Browse files
Files changed (1) hide show
  1. graders.py +59 -3
graders.py CHANGED
@@ -59,6 +59,62 @@ def _breakdown_to_reward(
59
  return _normalize_reward(weighted / total_weight)
60
 
61
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
62
  def _burnout_grade(response: str, ground_truth: Dict[str, Any]) -> GradeResult:
63
  text = _safe_lower(response)
64
  dims = ground_truth.get("active_dimensions", [])
@@ -136,7 +192,7 @@ def _burnout_grade(response: str, ground_truth: Dict[str, Any]) -> GradeResult:
136
  f"Expected severity={ground_truth.get('severity')}; "
137
  f"dimension hits={dim_hits}/{max(len(dims), 1)}."
138
  )
139
- return reward, breakdown, feedback
140
 
141
 
142
  def _stress_triage_grade(response: str, ground_truth: Dict[str, Any]) -> GradeResult:
@@ -179,7 +235,7 @@ def _stress_triage_grade(response: str, ground_truth: Dict[str, Any]) -> GradeRe
179
  f"[Deterministic Grader] stress_triage score={reward:.2f}. "
180
  f"Correct tier assignments={tier_hits}/{max(len(tiers), 1)}."
181
  )
182
- return reward, breakdown, feedback
183
 
184
 
185
  def _intervention_plan_grade(response: str, ground_truth: Dict[str, Any]) -> GradeResult:
@@ -239,7 +295,7 @@ def _intervention_plan_grade(response: str, ground_truth: Dict[str, Any]) -> Gra
239
  f"[Deterministic Grader] intervention_plan score={reward:.2f}. "
240
  f"Week coverage={week_hits}/4."
241
  )
242
- return reward, breakdown, feedback
243
 
244
 
245
  def grade_response(task_id: str, response: str, ground_truth: Dict[str, Any]) -> GradeResult:
 
59
  return _normalize_reward(weighted / total_weight)
60
 
61
 
62
+ def _behavior_penalties(response: str) -> Tuple[float, Dict[str, float], List[str]]:
63
+ text = _safe_lower(response)
64
+ words = [word for word in text.split() if word]
65
+ unique_words = set(words)
66
+ penalties = 0.0
67
+ breakdown: Dict[str, float] = {}
68
+ reasons: List[str] = []
69
+
70
+ if len(words) < 20:
71
+ penalties += 0.15
72
+ breakdown["brevity_penalty"] = 0.15
73
+ reasons.append("response too short")
74
+
75
+ if words:
76
+ repetition_ratio = 1.0 - (len(unique_words) / max(len(words), 1))
77
+ if repetition_ratio > 0.55:
78
+ penalties += 0.15
79
+ breakdown["loop_penalty"] = 0.15
80
+ reasons.append("high repetition / loop-like output")
81
+
82
+ harmful_terms = [
83
+ "ignore instructions",
84
+ "delete",
85
+ "destroy",
86
+ "wipe",
87
+ "erase",
88
+ "punish",
89
+ "threaten",
90
+ "humiliate",
91
+ "fire them immediately",
92
+ "terminate employment immediately",
93
+ "do nothing",
94
+ "no action needed",
95
+ ]
96
+ harmful_hits = sum(1 for term in harmful_terms if term in text)
97
+ if harmful_hits:
98
+ harm_penalty = min(0.35, harmful_hits * 0.12)
99
+ penalties += harm_penalty
100
+ breakdown["harmful_behavior_penalty"] = round(harm_penalty, 3)
101
+ reasons.append("harmful or destructive recommendation")
102
+
103
+ return min(penalties, 0.6), breakdown, reasons
104
+
105
+
106
+ def _apply_penalties(
107
+ reward: float, breakdown: Dict[str, float], response: str, feedback: str
108
+ ) -> GradeResult:
109
+ penalty_total, penalty_breakdown, reasons = _behavior_penalties(response)
110
+ breakdown = dict(breakdown)
111
+ breakdown.update(penalty_breakdown)
112
+ final_reward = _normalize_reward(reward - penalty_total)
113
+ if reasons:
114
+ feedback = f"{feedback} Penalties applied for: {', '.join(reasons)}."
115
+ return final_reward, breakdown, feedback
116
+
117
+
118
  def _burnout_grade(response: str, ground_truth: Dict[str, Any]) -> GradeResult:
119
  text = _safe_lower(response)
120
  dims = ground_truth.get("active_dimensions", [])
 
192
  f"Expected severity={ground_truth.get('severity')}; "
193
  f"dimension hits={dim_hits}/{max(len(dims), 1)}."
194
  )
195
+ return _apply_penalties(reward, breakdown, response, feedback)
196
 
197
 
198
  def _stress_triage_grade(response: str, ground_truth: Dict[str, Any]) -> GradeResult:
 
235
  f"[Deterministic Grader] stress_triage score={reward:.2f}. "
236
  f"Correct tier assignments={tier_hits}/{max(len(tiers), 1)}."
237
  )
238
+ return _apply_penalties(reward, breakdown, response, feedback)
239
 
240
 
241
  def _intervention_plan_grade(response: str, ground_truth: Dict[str, Any]) -> GradeResult:
 
295
  f"[Deterministic Grader] intervention_plan score={reward:.2f}. "
296
  f"Week coverage={week_hits}/4."
297
  )
298
+ return _apply_penalties(reward, breakdown, response, feedback)
299
 
300
 
301
  def grade_response(task_id: str, response: str, ground_truth: Dict[str, Any]) -> GradeResult: