chopratejas commited on
Commit
0302796
Β·
1 Parent(s): ef14c23

refactor: centralize error/priority pattern detection into error_detection module

Browse files

Extract duplicated regex patterns and error keywords from diff_compressor,
intelligent_context, search_compressor, smart_crusher, and text_compressor
into a shared headroom/transforms/error_detection.py module.

headroom/transforms/diff_compressor.py CHANGED
@@ -150,12 +150,10 @@ class DiffCompressor:
150
  _DELETED_FILE_MODE_PATTERN = re.compile(r"^deleted file mode")
151
  _RENAME_PATTERN = re.compile(r"^(rename|similarity|copy) ")
152
 
153
- # Priority patterns for context-aware hunk selection
154
- _PRIORITY_PATTERNS = [
155
- re.compile(r"\b(error|exception|fail|bug|fix)\b", re.IGNORECASE),
156
- re.compile(r"\b(todo|fixme|hack|xxx)\b", re.IGNORECASE),
157
- re.compile(r"\b(security|auth|password|secret|token)\b", re.IGNORECASE),
158
- ]
159
 
160
  def __init__(self, config: DiffCompressorConfig | None = None):
161
  """Initialize diff compressor.
 
150
  _DELETED_FILE_MODE_PATTERN = re.compile(r"^deleted file mode")
151
  _RENAME_PATTERN = re.compile(r"^(rename|similarity|copy) ")
152
 
153
+ # Priority patterns for context-aware hunk selection (centralized)
154
+ from headroom.transforms.error_detection import PRIORITY_PATTERNS_DIFF
155
+
156
+ _PRIORITY_PATTERNS = PRIORITY_PATTERNS_DIFF
 
 
157
 
158
  def __init__(self, config: DiffCompressorConfig | None = None):
159
  """Initialize diff compressor.
headroom/transforms/error_detection.py ADDED
@@ -0,0 +1,128 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Centralized error/importance detection for all transforms.
2
+
3
+ Design principle: Keywords serve as a FALLBACK safety net for error detection.
4
+ When TOIN field semantics are available, they take priority over keywords.
5
+
6
+ This module prevents each transform from maintaining its own hardcoded keyword
7
+ list, ensuring consistency and a single place to evolve detection logic.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import re
13
+
14
+ # ─── Canonical keyword sets ──────────────────────────────────────────────────
15
+ # These are the FALLBACK when TOIN semantics aren't available yet.
16
+ # They are intentionally broad to avoid missing errors.
17
+
18
+ ERROR_KEYWORDS: frozenset[str] = frozenset(
19
+ {
20
+ "error",
21
+ "exception",
22
+ "failed",
23
+ "failure",
24
+ "critical",
25
+ "fatal",
26
+ "crash",
27
+ "panic",
28
+ "abort",
29
+ "timeout",
30
+ "denied",
31
+ "rejected",
32
+ }
33
+ )
34
+
35
+ # Broader importance keywords (for line-level scoring, not item preservation)
36
+ IMPORTANCE_KEYWORDS: frozenset[str] = frozenset(
37
+ ERROR_KEYWORDS
38
+ | {
39
+ "warning",
40
+ "warn",
41
+ "todo",
42
+ "fixme",
43
+ "hack",
44
+ "xxx",
45
+ "bug",
46
+ "fix",
47
+ "important",
48
+ "note",
49
+ }
50
+ )
51
+
52
+ # Security-related keywords (for diff/search prioritization)
53
+ SECURITY_KEYWORDS: frozenset[str] = frozenset(
54
+ {
55
+ "security",
56
+ "auth",
57
+ "password",
58
+ "secret",
59
+ "token",
60
+ }
61
+ )
62
+
63
+ # ─── Compiled patterns (for line-level matching) ────────────────────────────
64
+ # Shared across text_compressor, diff_compressor, search_compressor
65
+
66
+ ERROR_PATTERN: re.Pattern[str] = re.compile(
67
+ r"\b(error|exception|fail(?:ed|ure)?|fatal|critical|crash|panic)\b",
68
+ re.IGNORECASE,
69
+ )
70
+
71
+ WARNING_PATTERN: re.Pattern[str] = re.compile(
72
+ r"\b(warn(?:ing)?)\b",
73
+ re.IGNORECASE,
74
+ )
75
+
76
+ IMPORTANCE_PATTERN: re.Pattern[str] = re.compile(
77
+ r"\b(important|note|todo|fixme|hack|xxx|bug|fix)\b",
78
+ re.IGNORECASE,
79
+ )
80
+
81
+ SECURITY_PATTERN: re.Pattern[str] = re.compile(
82
+ r"\b(security|auth|password|secret|token)\b",
83
+ re.IGNORECASE,
84
+ )
85
+
86
+ # Pre-built pattern lists for each compressor context
87
+ PRIORITY_PATTERNS_SEARCH: list[re.Pattern[str]] = [
88
+ ERROR_PATTERN,
89
+ WARNING_PATTERN,
90
+ IMPORTANCE_PATTERN,
91
+ ]
92
+
93
+ PRIORITY_PATTERNS_DIFF: list[re.Pattern[str]] = [
94
+ ERROR_PATTERN,
95
+ IMPORTANCE_PATTERN,
96
+ SECURITY_PATTERN,
97
+ ]
98
+
99
+ PRIORITY_PATTERNS_TEXT: list[re.Pattern[str]] = [
100
+ ERROR_PATTERN,
101
+ IMPORTANCE_PATTERN,
102
+ re.compile(r"^#+\s"), # Markdown headers
103
+ re.compile(r"^\*\*"), # Bold text
104
+ re.compile(r"^>\s"), # Quotes
105
+ ]
106
+
107
+ # ─── Quick check for message-level error indicators ─────────────────────────
108
+ # Used by intelligent_context.py for message signature creation
109
+
110
+ ERROR_INDICATOR_KEYWORDS: tuple[str, ...] = (
111
+ "error",
112
+ "fail",
113
+ "exception",
114
+ "traceback",
115
+ "fatal",
116
+ "panic",
117
+ "crash",
118
+ )
119
+
120
+
121
+ def content_has_error_indicators(text: str) -> bool:
122
+ """Check if text contains error indicators (fast keyword check).
123
+
124
+ Used for message signature creation and quick triage, NOT for
125
+ compression decisions (those should use TOIN when available).
126
+ """
127
+ text_lower = text.lower()
128
+ return any(kw in text_lower for kw in ERROR_INDICATOR_KEYWORDS)
headroom/transforms/intelligent_context.py CHANGED
@@ -79,12 +79,10 @@ def _create_message_signature(messages: list[dict[str, Any]]) -> Any:
79
  content = msg.get("content", "")
80
  if isinstance(content, str):
81
  total_content_length += len(content)
82
- # Check for error patterns (learned from TOIN, but basic heuristic here)
83
- content_lower = content.lower()
84
- if any(
85
- indicator in content_lower
86
- for indicator in ["error", "fail", "exception", "traceback"]
87
- ):
88
  has_error_indicators = True
89
  elif isinstance(content, list):
90
  # Anthropic format with content blocks
 
79
  content = msg.get("content", "")
80
  if isinstance(content, str):
81
  total_content_length += len(content)
82
+ # Check for error patterns (centralized, TOIN takes priority when available)
83
+ from headroom.transforms.error_detection import content_has_error_indicators
84
+
85
+ if content_has_error_indicators(content):
 
 
86
  has_error_indicators = True
87
  elif isinstance(content, list):
88
  # Anthropic format with content blocks
headroom/transforms/search_compressor.py CHANGED
@@ -87,12 +87,10 @@ class SearchCompressor:
87
  # Pattern for ripgrep with context (file-line-content or file:line:content)
88
  _RG_CONTEXT_PATTERN = re.compile(r"^([^:-]+)[:-](\d+)[:-](.*)$")
89
 
90
- # Error/important patterns to prioritize
91
- _PRIORITY_PATTERNS = [
92
- re.compile(r"\b(error|exception|fail|fatal)\b", re.IGNORECASE),
93
- re.compile(r"\b(warn|warning)\b", re.IGNORECASE),
94
- re.compile(r"\b(todo|fixme|hack|xxx)\b", re.IGNORECASE),
95
- ]
96
 
97
  def __init__(self, config: SearchCompressorConfig | None = None):
98
  """Initialize search compressor.
 
87
  # Pattern for ripgrep with context (file-line-content or file:line:content)
88
  _RG_CONTEXT_PATTERN = re.compile(r"^([^:-]+)[:-](\d+)[:-](.*)$")
89
 
90
+ # Error/important patterns to prioritize (centralized)
91
+ from headroom.transforms.error_detection import PRIORITY_PATTERNS_SEARCH
92
+
93
+ _PRIORITY_PATTERNS = PRIORITY_PATTERNS_SEARCH
 
 
94
 
95
  def __init__(self, config: SearchCompressorConfig | None = None):
96
  """Initialize search compressor.
headroom/transforms/smart_crusher.py CHANGED
@@ -77,6 +77,7 @@ from ..utils import (
77
  from .anchor_selector import AnchorSelector
78
  from .anchor_selector import DataPattern as AnchorDataPattern
79
  from .base import Transform
 
80
 
81
  logger = logging.getLogger(__name__)
82
 
@@ -623,22 +624,8 @@ def _detect_rare_status_values(items: list[dict], common_fields: set[str]) -> li
623
  # Error keywords for PRESERVATION guarantee (not crushability detection)
624
  # This is for the quality guarantee: "ALL error items are ALWAYS preserved"
625
  # regardless of how common they are. Used in _prioritize_indices().
626
- _ERROR_KEYWORDS_FOR_PRESERVATION = frozenset(
627
- {
628
- "error",
629
- "exception",
630
- "failed",
631
- "failure",
632
- "critical",
633
- "fatal",
634
- "crash",
635
- "panic",
636
- "abort",
637
- "timeout",
638
- "denied",
639
- "rejected",
640
- }
641
- )
642
 
643
 
644
  def _detect_error_items_for_preservation(items: list[dict]) -> list[int]:
 
77
  from .anchor_selector import AnchorSelector
78
  from .anchor_selector import DataPattern as AnchorDataPattern
79
  from .base import Transform
80
+ from .error_detection import ERROR_KEYWORDS
81
 
82
  logger = logging.getLogger(__name__)
83
 
 
624
  # Error keywords for PRESERVATION guarantee (not crushability detection)
625
  # This is for the quality guarantee: "ALL error items are ALWAYS preserved"
626
  # regardless of how common they are. Used in _prioritize_indices().
627
+ # Centralized in error_detection module for consistency across transforms.
628
+ _ERROR_KEYWORDS_FOR_PRESERVATION = ERROR_KEYWORDS
 
 
 
 
 
 
 
 
 
 
 
 
 
 
629
 
630
 
631
  def _detect_error_items_for_preservation(items: list[dict]) -> list[int]:
headroom/transforms/text_compressor.py CHANGED
@@ -13,7 +13,6 @@ Compression Strategy:
13
 
14
  from __future__ import annotations
15
 
16
- import re
17
  from dataclasses import dataclass, field
18
 
19
 
@@ -47,14 +46,10 @@ class TextCompressor:
47
  >>> print(result.compressed)
48
  """
49
 
50
- # Patterns that indicate important lines
51
- _IMPORTANT_PATTERNS = [
52
- re.compile(r"\b(error|exception|fail|warning)\b", re.IGNORECASE),
53
- re.compile(r"\b(important|note|todo|fixme)\b", re.IGNORECASE),
54
- re.compile(r"^#+\s"), # Markdown headers
55
- re.compile(r"^\*\*"), # Bold text
56
- re.compile(r"^>\s"), # Quotes
57
- ]
58
 
59
  def __init__(self, config: TextCompressorConfig | None = None):
60
  """Initialize text compressor.
 
13
 
14
  from __future__ import annotations
15
 
 
16
  from dataclasses import dataclass, field
17
 
18
 
 
46
  >>> print(result.compressed)
47
  """
48
 
49
+ # Patterns that indicate important lines (centralized in error_detection module)
50
+ from headroom.transforms.error_detection import PRIORITY_PATTERNS_TEXT
51
+
52
+ _IMPORTANT_PATTERNS = PRIORITY_PATTERNS_TEXT
 
 
 
 
53
 
54
  def __init__(self, config: TextCompressorConfig | None = None):
55
  """Initialize text compressor.