Spaces:
Build error
Build error
Commit Β·
0302796
1
Parent(s): ef14c23
refactor: centralize error/priority pattern detection into error_detection module
Browse filesExtract duplicated regex patterns and error keywords from diff_compressor,
intelligent_context, search_compressor, smart_crusher, and text_compressor
into a shared headroom/transforms/error_detection.py module.
headroom/transforms/diff_compressor.py
CHANGED
|
@@ -150,12 +150,10 @@ class DiffCompressor:
|
|
| 150 |
_DELETED_FILE_MODE_PATTERN = re.compile(r"^deleted file mode")
|
| 151 |
_RENAME_PATTERN = re.compile(r"^(rename|similarity|copy) ")
|
| 152 |
|
| 153 |
-
# Priority patterns for context-aware hunk selection
|
| 154 |
-
|
| 155 |
-
|
| 156 |
-
|
| 157 |
-
re.compile(r"\b(security|auth|password|secret|token)\b", re.IGNORECASE),
|
| 158 |
-
]
|
| 159 |
|
| 160 |
def __init__(self, config: DiffCompressorConfig | None = None):
|
| 161 |
"""Initialize diff compressor.
|
|
|
|
| 150 |
_DELETED_FILE_MODE_PATTERN = re.compile(r"^deleted file mode")
|
| 151 |
_RENAME_PATTERN = re.compile(r"^(rename|similarity|copy) ")
|
| 152 |
|
| 153 |
+
# Priority patterns for context-aware hunk selection (centralized)
|
| 154 |
+
from headroom.transforms.error_detection import PRIORITY_PATTERNS_DIFF
|
| 155 |
+
|
| 156 |
+
_PRIORITY_PATTERNS = PRIORITY_PATTERNS_DIFF
|
|
|
|
|
|
|
| 157 |
|
| 158 |
def __init__(self, config: DiffCompressorConfig | None = None):
|
| 159 |
"""Initialize diff compressor.
|
headroom/transforms/error_detection.py
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Centralized error/importance detection for all transforms.
|
| 2 |
+
|
| 3 |
+
Design principle: Keywords serve as a FALLBACK safety net for error detection.
|
| 4 |
+
When TOIN field semantics are available, they take priority over keywords.
|
| 5 |
+
|
| 6 |
+
This module prevents each transform from maintaining its own hardcoded keyword
|
| 7 |
+
list, ensuring consistency and a single place to evolve detection logic.
|
| 8 |
+
"""
|
| 9 |
+
|
| 10 |
+
from __future__ import annotations
|
| 11 |
+
|
| 12 |
+
import re
|
| 13 |
+
|
| 14 |
+
# βββ Canonical keyword sets ββββββββββββββββββββββββββββββββββββββββββββββββββ
|
| 15 |
+
# These are the FALLBACK when TOIN semantics aren't available yet.
|
| 16 |
+
# They are intentionally broad to avoid missing errors.
|
| 17 |
+
|
| 18 |
+
ERROR_KEYWORDS: frozenset[str] = frozenset(
|
| 19 |
+
{
|
| 20 |
+
"error",
|
| 21 |
+
"exception",
|
| 22 |
+
"failed",
|
| 23 |
+
"failure",
|
| 24 |
+
"critical",
|
| 25 |
+
"fatal",
|
| 26 |
+
"crash",
|
| 27 |
+
"panic",
|
| 28 |
+
"abort",
|
| 29 |
+
"timeout",
|
| 30 |
+
"denied",
|
| 31 |
+
"rejected",
|
| 32 |
+
}
|
| 33 |
+
)
|
| 34 |
+
|
| 35 |
+
# Broader importance keywords (for line-level scoring, not item preservation)
|
| 36 |
+
IMPORTANCE_KEYWORDS: frozenset[str] = frozenset(
|
| 37 |
+
ERROR_KEYWORDS
|
| 38 |
+
| {
|
| 39 |
+
"warning",
|
| 40 |
+
"warn",
|
| 41 |
+
"todo",
|
| 42 |
+
"fixme",
|
| 43 |
+
"hack",
|
| 44 |
+
"xxx",
|
| 45 |
+
"bug",
|
| 46 |
+
"fix",
|
| 47 |
+
"important",
|
| 48 |
+
"note",
|
| 49 |
+
}
|
| 50 |
+
)
|
| 51 |
+
|
| 52 |
+
# Security-related keywords (for diff/search prioritization)
|
| 53 |
+
SECURITY_KEYWORDS: frozenset[str] = frozenset(
|
| 54 |
+
{
|
| 55 |
+
"security",
|
| 56 |
+
"auth",
|
| 57 |
+
"password",
|
| 58 |
+
"secret",
|
| 59 |
+
"token",
|
| 60 |
+
}
|
| 61 |
+
)
|
| 62 |
+
|
| 63 |
+
# βββ Compiled patterns (for line-level matching) ββββββββββββββββββββββββββββ
|
| 64 |
+
# Shared across text_compressor, diff_compressor, search_compressor
|
| 65 |
+
|
| 66 |
+
ERROR_PATTERN: re.Pattern[str] = re.compile(
|
| 67 |
+
r"\b(error|exception|fail(?:ed|ure)?|fatal|critical|crash|panic)\b",
|
| 68 |
+
re.IGNORECASE,
|
| 69 |
+
)
|
| 70 |
+
|
| 71 |
+
WARNING_PATTERN: re.Pattern[str] = re.compile(
|
| 72 |
+
r"\b(warn(?:ing)?)\b",
|
| 73 |
+
re.IGNORECASE,
|
| 74 |
+
)
|
| 75 |
+
|
| 76 |
+
IMPORTANCE_PATTERN: re.Pattern[str] = re.compile(
|
| 77 |
+
r"\b(important|note|todo|fixme|hack|xxx|bug|fix)\b",
|
| 78 |
+
re.IGNORECASE,
|
| 79 |
+
)
|
| 80 |
+
|
| 81 |
+
SECURITY_PATTERN: re.Pattern[str] = re.compile(
|
| 82 |
+
r"\b(security|auth|password|secret|token)\b",
|
| 83 |
+
re.IGNORECASE,
|
| 84 |
+
)
|
| 85 |
+
|
| 86 |
+
# Pre-built pattern lists for each compressor context
|
| 87 |
+
PRIORITY_PATTERNS_SEARCH: list[re.Pattern[str]] = [
|
| 88 |
+
ERROR_PATTERN,
|
| 89 |
+
WARNING_PATTERN,
|
| 90 |
+
IMPORTANCE_PATTERN,
|
| 91 |
+
]
|
| 92 |
+
|
| 93 |
+
PRIORITY_PATTERNS_DIFF: list[re.Pattern[str]] = [
|
| 94 |
+
ERROR_PATTERN,
|
| 95 |
+
IMPORTANCE_PATTERN,
|
| 96 |
+
SECURITY_PATTERN,
|
| 97 |
+
]
|
| 98 |
+
|
| 99 |
+
PRIORITY_PATTERNS_TEXT: list[re.Pattern[str]] = [
|
| 100 |
+
ERROR_PATTERN,
|
| 101 |
+
IMPORTANCE_PATTERN,
|
| 102 |
+
re.compile(r"^#+\s"), # Markdown headers
|
| 103 |
+
re.compile(r"^\*\*"), # Bold text
|
| 104 |
+
re.compile(r"^>\s"), # Quotes
|
| 105 |
+
]
|
| 106 |
+
|
| 107 |
+
# βββ Quick check for message-level error indicators βββββββββββββββββββββββββ
|
| 108 |
+
# Used by intelligent_context.py for message signature creation
|
| 109 |
+
|
| 110 |
+
ERROR_INDICATOR_KEYWORDS: tuple[str, ...] = (
|
| 111 |
+
"error",
|
| 112 |
+
"fail",
|
| 113 |
+
"exception",
|
| 114 |
+
"traceback",
|
| 115 |
+
"fatal",
|
| 116 |
+
"panic",
|
| 117 |
+
"crash",
|
| 118 |
+
)
|
| 119 |
+
|
| 120 |
+
|
| 121 |
+
def content_has_error_indicators(text: str) -> bool:
|
| 122 |
+
"""Check if text contains error indicators (fast keyword check).
|
| 123 |
+
|
| 124 |
+
Used for message signature creation and quick triage, NOT for
|
| 125 |
+
compression decisions (those should use TOIN when available).
|
| 126 |
+
"""
|
| 127 |
+
text_lower = text.lower()
|
| 128 |
+
return any(kw in text_lower for kw in ERROR_INDICATOR_KEYWORDS)
|
headroom/transforms/intelligent_context.py
CHANGED
|
@@ -79,12 +79,10 @@ def _create_message_signature(messages: list[dict[str, Any]]) -> Any:
|
|
| 79 |
content = msg.get("content", "")
|
| 80 |
if isinstance(content, str):
|
| 81 |
total_content_length += len(content)
|
| 82 |
-
# Check for error patterns (
|
| 83 |
-
|
| 84 |
-
|
| 85 |
-
|
| 86 |
-
for indicator in ["error", "fail", "exception", "traceback"]
|
| 87 |
-
):
|
| 88 |
has_error_indicators = True
|
| 89 |
elif isinstance(content, list):
|
| 90 |
# Anthropic format with content blocks
|
|
|
|
| 79 |
content = msg.get("content", "")
|
| 80 |
if isinstance(content, str):
|
| 81 |
total_content_length += len(content)
|
| 82 |
+
# Check for error patterns (centralized, TOIN takes priority when available)
|
| 83 |
+
from headroom.transforms.error_detection import content_has_error_indicators
|
| 84 |
+
|
| 85 |
+
if content_has_error_indicators(content):
|
|
|
|
|
|
|
| 86 |
has_error_indicators = True
|
| 87 |
elif isinstance(content, list):
|
| 88 |
# Anthropic format with content blocks
|
headroom/transforms/search_compressor.py
CHANGED
|
@@ -87,12 +87,10 @@ class SearchCompressor:
|
|
| 87 |
# Pattern for ripgrep with context (file-line-content or file:line:content)
|
| 88 |
_RG_CONTEXT_PATTERN = re.compile(r"^([^:-]+)[:-](\d+)[:-](.*)$")
|
| 89 |
|
| 90 |
-
# Error/important patterns to prioritize
|
| 91 |
-
|
| 92 |
-
|
| 93 |
-
|
| 94 |
-
re.compile(r"\b(todo|fixme|hack|xxx)\b", re.IGNORECASE),
|
| 95 |
-
]
|
| 96 |
|
| 97 |
def __init__(self, config: SearchCompressorConfig | None = None):
|
| 98 |
"""Initialize search compressor.
|
|
|
|
| 87 |
# Pattern for ripgrep with context (file-line-content or file:line:content)
|
| 88 |
_RG_CONTEXT_PATTERN = re.compile(r"^([^:-]+)[:-](\d+)[:-](.*)$")
|
| 89 |
|
| 90 |
+
# Error/important patterns to prioritize (centralized)
|
| 91 |
+
from headroom.transforms.error_detection import PRIORITY_PATTERNS_SEARCH
|
| 92 |
+
|
| 93 |
+
_PRIORITY_PATTERNS = PRIORITY_PATTERNS_SEARCH
|
|
|
|
|
|
|
| 94 |
|
| 95 |
def __init__(self, config: SearchCompressorConfig | None = None):
|
| 96 |
"""Initialize search compressor.
|
headroom/transforms/smart_crusher.py
CHANGED
|
@@ -77,6 +77,7 @@ from ..utils import (
|
|
| 77 |
from .anchor_selector import AnchorSelector
|
| 78 |
from .anchor_selector import DataPattern as AnchorDataPattern
|
| 79 |
from .base import Transform
|
|
|
|
| 80 |
|
| 81 |
logger = logging.getLogger(__name__)
|
| 82 |
|
|
@@ -623,22 +624,8 @@ def _detect_rare_status_values(items: list[dict], common_fields: set[str]) -> li
|
|
| 623 |
# Error keywords for PRESERVATION guarantee (not crushability detection)
|
| 624 |
# This is for the quality guarantee: "ALL error items are ALWAYS preserved"
|
| 625 |
# regardless of how common they are. Used in _prioritize_indices().
|
| 626 |
-
|
| 627 |
-
|
| 628 |
-
"error",
|
| 629 |
-
"exception",
|
| 630 |
-
"failed",
|
| 631 |
-
"failure",
|
| 632 |
-
"critical",
|
| 633 |
-
"fatal",
|
| 634 |
-
"crash",
|
| 635 |
-
"panic",
|
| 636 |
-
"abort",
|
| 637 |
-
"timeout",
|
| 638 |
-
"denied",
|
| 639 |
-
"rejected",
|
| 640 |
-
}
|
| 641 |
-
)
|
| 642 |
|
| 643 |
|
| 644 |
def _detect_error_items_for_preservation(items: list[dict]) -> list[int]:
|
|
|
|
| 77 |
from .anchor_selector import AnchorSelector
|
| 78 |
from .anchor_selector import DataPattern as AnchorDataPattern
|
| 79 |
from .base import Transform
|
| 80 |
+
from .error_detection import ERROR_KEYWORDS
|
| 81 |
|
| 82 |
logger = logging.getLogger(__name__)
|
| 83 |
|
|
|
|
| 624 |
# Error keywords for PRESERVATION guarantee (not crushability detection)
|
| 625 |
# This is for the quality guarantee: "ALL error items are ALWAYS preserved"
|
| 626 |
# regardless of how common they are. Used in _prioritize_indices().
|
| 627 |
+
# Centralized in error_detection module for consistency across transforms.
|
| 628 |
+
_ERROR_KEYWORDS_FOR_PRESERVATION = ERROR_KEYWORDS
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 629 |
|
| 630 |
|
| 631 |
def _detect_error_items_for_preservation(items: list[dict]) -> list[int]:
|
headroom/transforms/text_compressor.py
CHANGED
|
@@ -13,7 +13,6 @@ Compression Strategy:
|
|
| 13 |
|
| 14 |
from __future__ import annotations
|
| 15 |
|
| 16 |
-
import re
|
| 17 |
from dataclasses import dataclass, field
|
| 18 |
|
| 19 |
|
|
@@ -47,14 +46,10 @@ class TextCompressor:
|
|
| 47 |
>>> print(result.compressed)
|
| 48 |
"""
|
| 49 |
|
| 50 |
-
# Patterns that indicate important lines
|
| 51 |
-
|
| 52 |
-
|
| 53 |
-
|
| 54 |
-
re.compile(r"^#+\s"), # Markdown headers
|
| 55 |
-
re.compile(r"^\*\*"), # Bold text
|
| 56 |
-
re.compile(r"^>\s"), # Quotes
|
| 57 |
-
]
|
| 58 |
|
| 59 |
def __init__(self, config: TextCompressorConfig | None = None):
|
| 60 |
"""Initialize text compressor.
|
|
|
|
| 13 |
|
| 14 |
from __future__ import annotations
|
| 15 |
|
|
|
|
| 16 |
from dataclasses import dataclass, field
|
| 17 |
|
| 18 |
|
|
|
|
| 46 |
>>> print(result.compressed)
|
| 47 |
"""
|
| 48 |
|
| 49 |
+
# Patterns that indicate important lines (centralized in error_detection module)
|
| 50 |
+
from headroom.transforms.error_detection import PRIORITY_PATTERNS_TEXT
|
| 51 |
+
|
| 52 |
+
_IMPORTANT_PATTERNS = PRIORITY_PATTERNS_TEXT
|
|
|
|
|
|
|
|
|
|
|
|
|
| 53 |
|
| 54 |
def __init__(self, config: TextCompressorConfig | None = None):
|
| 55 |
"""Initialize text compressor.
|