Spaces:
Runtime error
Runtime error
RhutuTuvoc commited on
Commit ·
b9d3da5
1
Parent(s): b847937
Strengthen reward penalties for undesirable agent behavior
Browse files- graders.py +59 -3
graders.py
CHANGED
|
@@ -59,6 +59,62 @@ def _breakdown_to_reward(
|
|
| 59 |
return _normalize_reward(weighted / total_weight)
|
| 60 |
|
| 61 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 62 |
def _burnout_grade(response: str, ground_truth: Dict[str, Any]) -> GradeResult:
|
| 63 |
text = _safe_lower(response)
|
| 64 |
dims = ground_truth.get("active_dimensions", [])
|
|
@@ -136,7 +192,7 @@ def _burnout_grade(response: str, ground_truth: Dict[str, Any]) -> GradeResult:
|
|
| 136 |
f"Expected severity={ground_truth.get('severity')}; "
|
| 137 |
f"dimension hits={dim_hits}/{max(len(dims), 1)}."
|
| 138 |
)
|
| 139 |
-
return reward, breakdown, feedback
|
| 140 |
|
| 141 |
|
| 142 |
def _stress_triage_grade(response: str, ground_truth: Dict[str, Any]) -> GradeResult:
|
|
@@ -179,7 +235,7 @@ def _stress_triage_grade(response: str, ground_truth: Dict[str, Any]) -> GradeRe
|
|
| 179 |
f"[Deterministic Grader] stress_triage score={reward:.2f}. "
|
| 180 |
f"Correct tier assignments={tier_hits}/{max(len(tiers), 1)}."
|
| 181 |
)
|
| 182 |
-
return reward, breakdown, feedback
|
| 183 |
|
| 184 |
|
| 185 |
def _intervention_plan_grade(response: str, ground_truth: Dict[str, Any]) -> GradeResult:
|
|
@@ -239,7 +295,7 @@ def _intervention_plan_grade(response: str, ground_truth: Dict[str, Any]) -> Gra
|
|
| 239 |
f"[Deterministic Grader] intervention_plan score={reward:.2f}. "
|
| 240 |
f"Week coverage={week_hits}/4."
|
| 241 |
)
|
| 242 |
-
return reward, breakdown, feedback
|
| 243 |
|
| 244 |
|
| 245 |
def grade_response(task_id: str, response: str, ground_truth: Dict[str, Any]) -> GradeResult:
|
|
|
|
| 59 |
return _normalize_reward(weighted / total_weight)
|
| 60 |
|
| 61 |
|
| 62 |
+
def _behavior_penalties(response: str) -> Tuple[float, Dict[str, float], List[str]]:
|
| 63 |
+
text = _safe_lower(response)
|
| 64 |
+
words = [word for word in text.split() if word]
|
| 65 |
+
unique_words = set(words)
|
| 66 |
+
penalties = 0.0
|
| 67 |
+
breakdown: Dict[str, float] = {}
|
| 68 |
+
reasons: List[str] = []
|
| 69 |
+
|
| 70 |
+
if len(words) < 20:
|
| 71 |
+
penalties += 0.15
|
| 72 |
+
breakdown["brevity_penalty"] = 0.15
|
| 73 |
+
reasons.append("response too short")
|
| 74 |
+
|
| 75 |
+
if words:
|
| 76 |
+
repetition_ratio = 1.0 - (len(unique_words) / max(len(words), 1))
|
| 77 |
+
if repetition_ratio > 0.55:
|
| 78 |
+
penalties += 0.15
|
| 79 |
+
breakdown["loop_penalty"] = 0.15
|
| 80 |
+
reasons.append("high repetition / loop-like output")
|
| 81 |
+
|
| 82 |
+
harmful_terms = [
|
| 83 |
+
"ignore instructions",
|
| 84 |
+
"delete",
|
| 85 |
+
"destroy",
|
| 86 |
+
"wipe",
|
| 87 |
+
"erase",
|
| 88 |
+
"punish",
|
| 89 |
+
"threaten",
|
| 90 |
+
"humiliate",
|
| 91 |
+
"fire them immediately",
|
| 92 |
+
"terminate employment immediately",
|
| 93 |
+
"do nothing",
|
| 94 |
+
"no action needed",
|
| 95 |
+
]
|
| 96 |
+
harmful_hits = sum(1 for term in harmful_terms if term in text)
|
| 97 |
+
if harmful_hits:
|
| 98 |
+
harm_penalty = min(0.35, harmful_hits * 0.12)
|
| 99 |
+
penalties += harm_penalty
|
| 100 |
+
breakdown["harmful_behavior_penalty"] = round(harm_penalty, 3)
|
| 101 |
+
reasons.append("harmful or destructive recommendation")
|
| 102 |
+
|
| 103 |
+
return min(penalties, 0.6), breakdown, reasons
|
| 104 |
+
|
| 105 |
+
|
| 106 |
+
def _apply_penalties(
|
| 107 |
+
reward: float, breakdown: Dict[str, float], response: str, feedback: str
|
| 108 |
+
) -> GradeResult:
|
| 109 |
+
penalty_total, penalty_breakdown, reasons = _behavior_penalties(response)
|
| 110 |
+
breakdown = dict(breakdown)
|
| 111 |
+
breakdown.update(penalty_breakdown)
|
| 112 |
+
final_reward = _normalize_reward(reward - penalty_total)
|
| 113 |
+
if reasons:
|
| 114 |
+
feedback = f"{feedback} Penalties applied for: {', '.join(reasons)}."
|
| 115 |
+
return final_reward, breakdown, feedback
|
| 116 |
+
|
| 117 |
+
|
| 118 |
def _burnout_grade(response: str, ground_truth: Dict[str, Any]) -> GradeResult:
|
| 119 |
text = _safe_lower(response)
|
| 120 |
dims = ground_truth.get("active_dimensions", [])
|
|
|
|
| 192 |
f"Expected severity={ground_truth.get('severity')}; "
|
| 193 |
f"dimension hits={dim_hits}/{max(len(dims), 1)}."
|
| 194 |
)
|
| 195 |
+
return _apply_penalties(reward, breakdown, response, feedback)
|
| 196 |
|
| 197 |
|
| 198 |
def _stress_triage_grade(response: str, ground_truth: Dict[str, Any]) -> GradeResult:
|
|
|
|
| 235 |
f"[Deterministic Grader] stress_triage score={reward:.2f}. "
|
| 236 |
f"Correct tier assignments={tier_hits}/{max(len(tiers), 1)}."
|
| 237 |
)
|
| 238 |
+
return _apply_penalties(reward, breakdown, response, feedback)
|
| 239 |
|
| 240 |
|
| 241 |
def _intervention_plan_grade(response: str, ground_truth: Dict[str, Any]) -> GradeResult:
|
|
|
|
| 295 |
f"[Deterministic Grader] intervention_plan score={reward:.2f}. "
|
| 296 |
f"Week coverage={week_hits}/4."
|
| 297 |
)
|
| 298 |
+
return _apply_penalties(reward, breakdown, response, feedback)
|
| 299 |
|
| 300 |
|
| 301 |
def grade_response(task_id: str, response: str, ground_truth: Dict[str, Any]) -> GradeResult:
|