chopratejas commited on
Commit
80f8f0d
Β·
1 Parent(s): a7975c5

Fix proxy crash when torch not installed (kompress lazy imports)

Browse files

The proxy startup crashed with `ModuleNotFoundError: No module named
'torch'` when installed with just `[proxy]` extras because
kompress_compressor.py had unconditional top-level torch imports.
Moved torch/transformers imports to be lazy so the module is safely
importable without the [ml] extra. Added tests for import safety.
Bumped version to 0.4.5

headroom/__init__.py CHANGED
@@ -153,7 +153,7 @@ from .transforms import (
153
  TransformPipeline,
154
  )
155
 
156
- __version__ = "0.4.4"
157
 
158
  __all__ = [
159
  # Main client
 
153
  TransformPipeline,
154
  )
155
 
156
+ __version__ = "0.4.5"
157
 
158
  __all__ = [
159
  # Main client
headroom/transforms/kompress_compressor.py CHANGED
@@ -3,8 +3,7 @@
3
  Drop-in replacement for LLMLingua-2. Auto-downloads the model from
4
  HuggingFace (chopratejas/kompress-base) on first use.
5
 
6
- No extra pip install needed β€” uses transformers + safetensors
7
- which are already Headroom dependencies.
8
 
9
  Usage:
10
  >>> from headroom.transforms.kompress_compressor import KompressCompressor
@@ -20,10 +19,6 @@ import threading
20
  from dataclasses import dataclass
21
  from typing import Any
22
 
23
- import torch
24
- import torch.nn as nn
25
- from transformers import AutoModel, AutoTokenizer
26
-
27
  from ..config import TransformResult
28
  from ..tokenizer import Tokenizer
29
  from .base import Transform
@@ -39,64 +34,95 @@ _kompress_tokenizer = None
39
  _kompress_lock = threading.Lock()
40
 
41
 
42
- # ── Model Architecture (must match training) ──────────────────────────
43
-
 
 
 
 
 
44
 
45
- class HeadroomCompressorModel(nn.Module):
46
- """Dual-head ModernBERT: token classification + span importance CNN."""
 
47
 
48
- def __init__(self, model_name: str = "answerdotai/ModernBERT-base"):
49
- super().__init__()
50
- self.encoder = AutoModel.from_pretrained(model_name, attn_implementation="eager")
51
- hidden_size = self.encoder.config.hidden_size # 768
52
 
53
- # Head 1: Token keep/discard
54
- self.token_dropout = nn.Dropout(0.1)
55
- self.token_head = nn.Linear(hidden_size, 2)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
56
 
57
- # Head 2: Span importance (1D CNN)
58
- self.span_conv = nn.Sequential(
59
- nn.Conv1d(hidden_size, 256, kernel_size=5, padding=2),
60
- nn.GELU(),
61
- nn.Conv1d(256, 1, kernel_size=3, padding=1),
62
- nn.Sigmoid(),
63
- )
64
 
65
- def get_keep_mask(self, input_ids: torch.Tensor, attention_mask: torch.Tensor) -> torch.Tensor:
66
- """Get per-token keep/discard decision. True = keep."""
67
- with torch.no_grad():
68
- hidden = self.encoder(input_ids, attention_mask=attention_mask).last_hidden_state
 
69
 
70
- # Token head: binary classifier β€” argmax decides keep/discard
71
- token_logits = self.token_head(hidden) # [B, L, 2]
72
- token_keep = token_logits[:, :, 1] > token_logits[:, :, 0] # True if class 1 > class 0
 
73
 
74
- # Span head: boost tokens in important spans
75
- # If a token is borderline but its span is important, keep it
76
- span_scores = self.span_conv(hidden.transpose(1, 2)).squeeze(1)
77
- span_boost = span_scores > 0.5 # span says this region matters
78
 
79
- # Keep if: token head says keep, OR token is borderline and span says keep
80
- token_probs = torch.softmax(token_logits, dim=-1)[:, :, 1]
81
- borderline = (token_probs > 0.3) & (token_probs <= 0.5)
82
- keep = token_keep | (borderline & span_boost)
83
 
84
- return keep # type: ignore[no-any-return]
 
 
 
 
 
 
85
 
86
- def get_scores(self, input_ids: torch.Tensor, attention_mask: torch.Tensor) -> torch.Tensor:
87
- """Get per-token importance scores (for ranking when target_ratio is set)."""
88
- with torch.no_grad():
89
- hidden = self.encoder(input_ids, attention_mask=attention_mask).last_hidden_state
90
- token_probs = torch.softmax(self.token_head(hidden), dim=-1)[:, :, 1]
91
- span_scores = self.span_conv(hidden.transpose(1, 2)).squeeze(1)
92
- return token_probs * (0.5 + 0.5 * span_scores) # type: ignore[no-any-return]
93
 
94
 
95
  # ── Model Loading ─────────────────────────────────────────────────────
96
 
97
 
98
- def _load_kompress(device: str = "auto") -> tuple[HeadroomCompressorModel, Any]:
99
  """Download from HuggingFace and load the Kompress model."""
 
 
 
100
  global _kompress_model, _kompress_tokenizer
101
 
102
  with _kompress_lock:
@@ -111,6 +137,7 @@ def _load_kompress(device: str = "auto") -> tuple[HeadroomCompressorModel, Any]:
111
  weights_path = hf_hub_download(HF_MODEL_ID, "model.safetensors")
112
 
113
  # Load architecture
 
114
  model = HeadroomCompressorModel()
115
 
116
  # Load trained weights
@@ -139,19 +166,6 @@ def _load_kompress(device: str = "auto") -> tuple[HeadroomCompressorModel, Any]:
139
  return model, tokenizer
140
 
141
 
142
- def is_kompress_available() -> bool:
143
- """Check if Kompress dependencies are available (requires [ml] extra)."""
144
- try:
145
- import huggingface_hub # noqa: F401
146
- import safetensors # noqa: F401
147
- import torch # noqa: F401
148
- import transformers # noqa: F401
149
-
150
- return True
151
- except ImportError:
152
- return False
153
-
154
-
155
  def unload_kompress_model() -> bool:
156
  """Unload the Kompress model to free memory."""
157
  global _kompress_model, _kompress_tokenizer
@@ -159,8 +173,13 @@ def unload_kompress_model() -> bool:
159
  if _kompress_model is not None:
160
  _kompress_model = None
161
  _kompress_tokenizer = None
162
- if torch.cuda.is_available():
163
- torch.cuda.empty_cache()
 
 
 
 
 
164
  return True
165
  return False
166
 
 
3
  Drop-in replacement for LLMLingua-2. Auto-downloads the model from
4
  HuggingFace (chopratejas/kompress-base) on first use.
5
 
6
+ Requires the [ml] extra: pip install headroom-ai[ml]
 
7
 
8
  Usage:
9
  >>> from headroom.transforms.kompress_compressor import KompressCompressor
 
19
  from dataclasses import dataclass
20
  from typing import Any
21
 
 
 
 
 
22
  from ..config import TransformResult
23
  from ..tokenizer import Tokenizer
24
  from .base import Transform
 
34
  _kompress_lock = threading.Lock()
35
 
36
 
37
+ def is_kompress_available() -> bool:
38
+ """Check if Kompress dependencies are available (requires [ml] extra)."""
39
+ try:
40
+ import huggingface_hub # noqa: F401
41
+ import safetensors # noqa: F401
42
+ import torch # noqa: F401
43
+ import transformers # noqa: F401
44
 
45
+ return True
46
+ except ImportError:
47
+ return False
48
 
 
 
 
 
49
 
50
+ # ── Model Architecture (must match training) ──────────────────────────
51
+ # torch/transformers are imported lazily β€” only when actually needed.
52
+ # This allows `from kompress_compressor import is_kompress_available`
53
+ # to work without torch installed.
54
+
55
+
56
+ def _get_model_class() -> type:
57
+ """Return the HeadroomCompressorModel class, importing torch on demand."""
58
+ import torch
59
+ import torch.nn as nn
60
+ from transformers import AutoModel
61
+
62
+ class HeadroomCompressorModel(nn.Module):
63
+ """Dual-head ModernBERT: token classification + span importance CNN."""
64
+
65
+ def __init__(self, model_name: str = "answerdotai/ModernBERT-base"):
66
+ super().__init__()
67
+ self.encoder = AutoModel.from_pretrained(model_name, attn_implementation="eager")
68
+ hidden_size = self.encoder.config.hidden_size # 768
69
+
70
+ # Head 1: Token keep/discard
71
+ self.token_dropout = nn.Dropout(0.1)
72
+ self.token_head = nn.Linear(hidden_size, 2)
73
+
74
+ # Head 2: Span importance (1D CNN)
75
+ self.span_conv = nn.Sequential(
76
+ nn.Conv1d(hidden_size, 256, kernel_size=5, padding=2),
77
+ nn.GELU(),
78
+ nn.Conv1d(256, 1, kernel_size=3, padding=1),
79
+ nn.Sigmoid(),
80
+ )
81
 
82
+ def get_keep_mask(
83
+ self, input_ids: torch.Tensor, attention_mask: torch.Tensor
84
+ ) -> torch.Tensor:
85
+ """Get per-token keep/discard decision. True = keep."""
86
+ with torch.no_grad():
87
+ hidden = self.encoder(input_ids, attention_mask=attention_mask).last_hidden_state
 
88
 
89
+ # Token head: binary classifier β€” argmax decides keep/discard
90
+ token_logits = self.token_head(hidden) # [B, L, 2]
91
+ token_keep = (
92
+ token_logits[:, :, 1] > token_logits[:, :, 0]
93
+ ) # True if class 1 > class 0
94
 
95
+ # Span head: boost tokens in important spans
96
+ # If a token is borderline but its span is important, keep it
97
+ span_scores = self.span_conv(hidden.transpose(1, 2)).squeeze(1)
98
+ span_boost = span_scores > 0.5 # span says this region matters
99
 
100
+ # Keep if: token head says keep, OR token is borderline and span says keep
101
+ token_probs = torch.softmax(token_logits, dim=-1)[:, :, 1]
102
+ borderline = (token_probs > 0.3) & (token_probs <= 0.5)
103
+ keep = token_keep | (borderline & span_boost)
104
 
105
+ return keep # type: ignore[no-any-return]
 
 
 
106
 
107
+ def get_scores(self, input_ids: torch.Tensor, attention_mask: torch.Tensor) -> torch.Tensor:
108
+ """Get per-token importance scores (for ranking when target_ratio is set)."""
109
+ with torch.no_grad():
110
+ hidden = self.encoder(input_ids, attention_mask=attention_mask).last_hidden_state
111
+ token_probs = torch.softmax(self.token_head(hidden), dim=-1)[:, :, 1]
112
+ span_scores = self.span_conv(hidden.transpose(1, 2)).squeeze(1)
113
+ return token_probs * (0.5 + 0.5 * span_scores) # type: ignore[no-any-return]
114
 
115
+ return HeadroomCompressorModel
 
 
 
 
 
 
116
 
117
 
118
  # ── Model Loading ─────────────────────────────────────────────────────
119
 
120
 
121
+ def _load_kompress(device: str = "auto") -> tuple[Any, Any]:
122
  """Download from HuggingFace and load the Kompress model."""
123
+ import torch
124
+ from transformers import AutoTokenizer
125
+
126
  global _kompress_model, _kompress_tokenizer
127
 
128
  with _kompress_lock:
 
137
  weights_path = hf_hub_download(HF_MODEL_ID, "model.safetensors")
138
 
139
  # Load architecture
140
+ HeadroomCompressorModel = _get_model_class()
141
  model = HeadroomCompressorModel()
142
 
143
  # Load trained weights
 
166
  return model, tokenizer
167
 
168
 
 
 
 
 
 
 
 
 
 
 
 
 
 
169
  def unload_kompress_model() -> bool:
170
  """Unload the Kompress model to free memory."""
171
  global _kompress_model, _kompress_tokenizer
 
173
  if _kompress_model is not None:
174
  _kompress_model = None
175
  _kompress_tokenizer = None
176
+ try:
177
+ import torch
178
+
179
+ if torch.cuda.is_available():
180
+ torch.cuda.empty_cache()
181
+ except ImportError:
182
+ pass
183
  return True
184
  return False
185
 
pyproject.toml CHANGED
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
 
5
  [project]
6
  name = "headroom-ai"
7
- version = "0.4.4"
8
  description = "The Context Optimization Layer for LLM Applications - Cut costs by 50-90%"
9
  readme = "README.md"
10
  license = "Apache-2.0"
 
4
 
5
  [project]
6
  name = "headroom-ai"
7
+ version = "0.4.5"
8
  description = "The Context Optimization Layer for LLM Applications - Cut costs by 50-90%"
9
  readme = "README.md"
10
  license = "Apache-2.0"
tests/test_transforms/test_kompress_compressor.py ADDED
@@ -0,0 +1,204 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Tests for Kompress compressor.
2
+
3
+ Covers:
4
+ - Lazy imports: module importable without torch installed
5
+ - is_kompress_available(): correct detection of [ml] extra
6
+ - KompressConfig / KompressResult: dataclass defaults
7
+ - KompressCompressor: passthrough for short content, fallback on error
8
+ - Transform interface: apply() method
9
+ """
10
+
11
+ from unittest.mock import MagicMock, patch
12
+
13
+ # ── Import safety (the whole point of the fix) ─────────────────────────
14
+
15
+
16
+ class TestLazyImports:
17
+ """The module must be importable without torch/transformers."""
18
+
19
+ def test_is_kompress_available_importable(self) -> None:
20
+ """is_kompress_available can be imported even without torch."""
21
+ from headroom.transforms.kompress_compressor import is_kompress_available
22
+
23
+ # Should return bool (True or False depending on environment)
24
+ result = is_kompress_available()
25
+ assert isinstance(result, bool)
26
+
27
+ def test_module_import_without_torch(self) -> None:
28
+ """Importing the module with torch blocked should not raise."""
29
+ import sys
30
+
31
+ # Block torch imports
32
+ with patch.dict(sys.modules, {"torch": None, "torch.nn": None}):
33
+ # Force re-evaluation of is_kompress_available
34
+ from headroom.transforms.kompress_compressor import is_kompress_available
35
+
36
+ # Should gracefully return False, not crash
37
+ assert is_kompress_available() is False
38
+
39
+ def test_dataclasses_importable_without_torch(self) -> None:
40
+ """KompressConfig, KompressResult, KompressCompressor are importable without torch."""
41
+ from headroom.transforms.kompress_compressor import (
42
+ KompressCompressor, # noqa: F401
43
+ KompressConfig,
44
+ KompressResult,
45
+ )
46
+
47
+ # These don't need torch to instantiate
48
+ config = KompressConfig()
49
+ assert config.device == "auto"
50
+ assert config.enable_ccr is True
51
+
52
+ result = KompressResult(
53
+ compressed="hello",
54
+ original="hello world",
55
+ original_tokens=2,
56
+ compressed_tokens=1,
57
+ compression_ratio=0.5,
58
+ )
59
+ assert result.tokens_saved == 1
60
+ assert result.savings_percentage == 50.0
61
+
62
+
63
+ # ── KompressResult ──────────────────────────────────────────────────────
64
+
65
+
66
+ class TestKompressResult:
67
+ def test_tokens_saved(self) -> None:
68
+ from headroom.transforms.kompress_compressor import KompressResult
69
+
70
+ r = KompressResult(
71
+ compressed="a b",
72
+ original="a b c d",
73
+ original_tokens=4,
74
+ compressed_tokens=2,
75
+ compression_ratio=0.5,
76
+ )
77
+ assert r.tokens_saved == 2
78
+
79
+ def test_tokens_saved_no_negative(self) -> None:
80
+ from headroom.transforms.kompress_compressor import KompressResult
81
+
82
+ r = KompressResult(
83
+ compressed="a b c d e",
84
+ original="a b c",
85
+ original_tokens=3,
86
+ compressed_tokens=5,
87
+ compression_ratio=1.67,
88
+ )
89
+ assert r.tokens_saved == 0
90
+
91
+ def test_savings_percentage_zero_tokens(self) -> None:
92
+ from headroom.transforms.kompress_compressor import KompressResult
93
+
94
+ r = KompressResult(
95
+ compressed="",
96
+ original="",
97
+ original_tokens=0,
98
+ compressed_tokens=0,
99
+ compression_ratio=1.0,
100
+ )
101
+ assert r.savings_percentage == 0.0
102
+
103
+ def test_default_model(self) -> None:
104
+ from headroom.transforms.kompress_compressor import HF_MODEL_ID, KompressResult
105
+
106
+ r = KompressResult(
107
+ compressed="x",
108
+ original="x y",
109
+ original_tokens=2,
110
+ compressed_tokens=1,
111
+ compression_ratio=0.5,
112
+ )
113
+ assert r.model_used == HF_MODEL_ID
114
+
115
+
116
+ # ── KompressCompressor (without model) ──────────────────────────────────
117
+
118
+
119
+ class TestKompressCompressorPassthrough:
120
+ """Test compressor behavior that doesn't require the actual model."""
121
+
122
+ def test_short_content_passthrough(self) -> None:
123
+ """Content under 10 words should pass through unchanged."""
124
+ from headroom.transforms.kompress_compressor import KompressCompressor
125
+
126
+ compressor = KompressCompressor()
127
+ result = compressor.compress("hello world")
128
+ assert result.compressed == "hello world"
129
+ assert result.compression_ratio == 1.0
130
+ assert result.original_tokens == 2
131
+ assert result.compressed_tokens == 2
132
+
133
+ def test_empty_content_passthrough(self) -> None:
134
+ from headroom.transforms.kompress_compressor import KompressCompressor
135
+
136
+ compressor = KompressCompressor()
137
+ result = compressor.compress("")
138
+ assert result.compressed == ""
139
+ assert result.compression_ratio == 1.0
140
+
141
+ def test_fallback_on_model_error(self) -> None:
142
+ """If _load_kompress fails, compress should return passthrough."""
143
+ from headroom.transforms.kompress_compressor import KompressCompressor
144
+
145
+ compressor = KompressCompressor()
146
+ long_text = " ".join(f"word{i}" for i in range(20))
147
+
148
+ with patch(
149
+ "headroom.transforms.kompress_compressor._load_kompress",
150
+ side_effect=RuntimeError("no model"),
151
+ ):
152
+ result = compressor.compress(long_text)
153
+ assert result.compressed == long_text
154
+ assert result.compression_ratio == 1.0
155
+
156
+
157
+ # ── Transform interface ─────────────────────────────────────────────────
158
+
159
+
160
+ class TestKompressTransformInterface:
161
+ def test_apply_short_messages_unchanged(self) -> None:
162
+ """Messages with <10 words should pass through apply() unchanged."""
163
+ from headroom.transforms.kompress_compressor import KompressCompressor
164
+
165
+ compressor = KompressCompressor()
166
+ messages = [
167
+ {"role": "user", "content": "hello"},
168
+ {"role": "tool", "content": "short"},
169
+ ]
170
+ tokenizer = MagicMock()
171
+ tokenizer.count_text = MagicMock(return_value=5)
172
+
173
+ result = compressor.apply(messages, tokenizer)
174
+ assert len(result.messages) == 2
175
+ assert result.messages[0]["content"] == "hello"
176
+ assert result.messages[1]["content"] == "short"
177
+
178
+ def test_apply_preserves_user_messages(self) -> None:
179
+ """User messages should never be compressed."""
180
+ from headroom.transforms.kompress_compressor import KompressCompressor
181
+
182
+ compressor = KompressCompressor()
183
+ long_text = " ".join(f"word{i}" for i in range(50))
184
+ messages = [{"role": "user", "content": long_text}]
185
+ tokenizer = MagicMock()
186
+ tokenizer.count_text = MagicMock(return_value=50)
187
+
188
+ with patch(
189
+ "headroom.transforms.kompress_compressor._load_kompress",
190
+ side_effect=RuntimeError("should not be called"),
191
+ ):
192
+ result = compressor.apply(messages, tokenizer)
193
+ assert result.messages[0]["content"] == long_text
194
+
195
+
196
+ # ── unload_kompress_model ───────────────────────────────────────────────
197
+
198
+
199
+ class TestUnloadKompressModel:
200
+ def test_unload_when_no_model(self) -> None:
201
+ from headroom.transforms.kompress_compressor import unload_kompress_model
202
+
203
+ # Should return False when no model is loaded
204
+ assert unload_kompress_model() is False