angpt commited on
Commit
c106272
·
1 Parent(s): 3b38650

Adding any-llm as a backed provider

Browse files
headroom/backends/__init__.py CHANGED
@@ -3,17 +3,20 @@
3
  Backends handle the translation between the proxy's canonical format
4
  (Anthropic Messages API) and provider-specific APIs.
5
 
6
- Uses LiteLLM for broad provider support:
7
- - bedrock: AWS Bedrock (Claude, Cohere, Mistral, etc.)
8
- - vertex_ai: Google Vertex AI (Claude, Gemini, etc.)
9
- - azure: Azure OpenAI (GPT-4, etc.)
10
- - And 100+ more providers...
11
 
12
  Usage:
 
13
  headroom proxy --backend litellm-bedrock --region us-west-2
 
 
 
14
  """
15
 
 
16
  from .base import Backend, BackendResponse, StreamEvent
17
  from .litellm import LiteLLMBackend
18
 
19
- __all__ = ["Backend", "BackendResponse", "StreamEvent", "LiteLLMBackend"]
 
3
  Backends handle the translation between the proxy's canonical format
4
  (Anthropic Messages API) and provider-specific APIs.
5
 
6
+ Supported backend libraries:
7
+ - LiteLLM: 100+ providers (bedrock, vertex_ai, azure, openrouter, etc.)
8
+ - any-llm: 38+ providers (openai, anthropic, mistral, groq, ollama, etc.)
 
 
9
 
10
  Usage:
11
+ # LiteLLM backend
12
  headroom proxy --backend litellm-bedrock --region us-west-2
13
+
14
+ # any-llm backend
15
+ headroom proxy --backend anyllm --anyllm-provider openai
16
  """
17
 
18
+ from .anyllm import AnyLLMBackend
19
  from .base import Backend, BackendResponse, StreamEvent
20
  from .litellm import LiteLLMBackend
21
 
22
+ __all__ = ["Backend", "BackendResponse", "StreamEvent", "LiteLLMBackend", "AnyLLMBackend"]
headroom/backends/anyllm.py ADDED
@@ -0,0 +1,476 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """any-llm backend for Headroom.
2
+
3
+ Talk to 38+ LLM providers (OpenAI, Mistral, Groq, Ollama, Bedrock, etc.)
4
+ through a single interface. Auth and format translation handled automatically.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import logging
10
+ import uuid
11
+ from collections.abc import AsyncIterator
12
+ from typing import Any
13
+
14
+ from .base import Backend, BackendResponse, StreamEvent
15
+
16
+ logger = logging.getLogger(__name__)
17
+
18
+ try:
19
+ from any_llm import acompletion
20
+
21
+ ANYLLM_AVAILABLE = True
22
+ except ImportError:
23
+ ANYLLM_AVAILABLE = False
24
+ acompletion = None # type: ignore
25
+
26
+
27
+ class AnyLLMBackend(Backend):
28
+ """Backend using any-llm for multi-provider support."""
29
+
30
+ def __init__(
31
+ self,
32
+ provider: str = "openai",
33
+ api_key: str | None = None,
34
+ api_base: str | None = None,
35
+ ):
36
+ if not ANYLLM_AVAILABLE:
37
+ raise ImportError(
38
+ "any-llm-sdk is required for AnyLLMBackend. "
39
+ "Install with: pip install 'any-llm-sdk[all]'"
40
+ )
41
+
42
+ self.provider = provider.lower()
43
+ self.api_key = api_key
44
+ self.api_base = api_base
45
+
46
+ logger.info(f"any-llm backend initialized (provider={provider})")
47
+
48
+ @property
49
+ def name(self) -> str:
50
+ return f"anyllm-{self.provider}"
51
+
52
+ def map_model_id(self, model: str) -> str:
53
+ """Pass through model name - any-llm handles provider-specific naming."""
54
+ return model
55
+
56
+ def supports_model(self, model: str) -> bool:
57
+ """any-llm supports any model the provider supports."""
58
+ return True
59
+
60
+ def _convert_messages_for_anyllm(self, messages: list[dict[str, Any]]) -> list[dict[str, Any]]:
61
+ """Convert Anthropic message format to OpenAI/any-llm format."""
62
+ converted = []
63
+ for msg in messages:
64
+ role = msg.get("role", "user")
65
+ content = msg.get("content", "")
66
+
67
+ if isinstance(content, str):
68
+ converted.append({"role": role, "content": content})
69
+ continue
70
+
71
+ if isinstance(content, list):
72
+ text_parts = []
73
+ has_complex_content = False
74
+
75
+ for block in content:
76
+ if isinstance(block, dict):
77
+ if block.get("type") == "text":
78
+ text_parts.append(block.get("text", ""))
79
+ elif block.get("type") in ("tool_use", "tool_result", "image"):
80
+ has_complex_content = True
81
+ break
82
+
83
+ if not has_complex_content and text_parts:
84
+ converted.append({"role": role, "content": "\n".join(text_parts)})
85
+ else:
86
+ openai_content = self._convert_content_blocks(content)
87
+ converted.append({"role": role, "content": openai_content})
88
+
89
+ return converted
90
+
91
+ def _convert_content_blocks(self, blocks: list[dict[str, Any]]) -> list[dict[str, Any]] | str:
92
+ """Convert Anthropic content blocks to OpenAI format."""
93
+ openai_blocks = []
94
+
95
+ for block in blocks:
96
+ block_type = block.get("type")
97
+
98
+ if block_type == "text":
99
+ openai_blocks.append({"type": "text", "text": block.get("text", "")})
100
+
101
+ elif block_type == "image":
102
+ source = block.get("source", {})
103
+ if source.get("type") == "base64":
104
+ media_type = source.get("media_type", "image/png")
105
+ data = source.get("data", "")
106
+ openai_blocks.append(
107
+ {
108
+ "type": "image_url",
109
+ "image_url": {"url": f"data:{media_type};base64,{data}"},
110
+ }
111
+ )
112
+ elif source.get("type") == "url":
113
+ openai_blocks.append(
114
+ {"type": "image_url", "image_url": {"url": source.get("url", "")}}
115
+ )
116
+
117
+ if len(openai_blocks) == 1 and openai_blocks[0].get("type") == "text":
118
+ return openai_blocks[0]["text"]
119
+
120
+ return openai_blocks if openai_blocks else ""
121
+
122
+ def _to_anthropic_response(
123
+ self,
124
+ anyllm_response: Any,
125
+ original_model: str,
126
+ ) -> dict[str, Any]:
127
+ """Convert any-llm/OpenAI response to Anthropic format."""
128
+ msg_id = f"msg_{uuid.uuid4().hex[:24]}"
129
+
130
+ choice = anyllm_response.choices[0]
131
+ message = choice.message
132
+
133
+ content = []
134
+ if message.content:
135
+ content.append({"type": "text", "text": message.content})
136
+
137
+ if hasattr(message, "tool_calls") and message.tool_calls:
138
+ for tc in message.tool_calls:
139
+ content.append(
140
+ {
141
+ "type": "tool_use",
142
+ "id": tc.id,
143
+ "name": tc.function.name,
144
+ "input": tc.function.arguments,
145
+ }
146
+ )
147
+
148
+ stop_reason_map = {
149
+ "stop": "end_turn",
150
+ "length": "max_tokens",
151
+ "tool_calls": "tool_use",
152
+ "content_filter": "end_turn",
153
+ }
154
+ finish_reason = getattr(choice, "finish_reason", "stop") or "stop"
155
+ stop_reason = stop_reason_map.get(finish_reason, "end_turn")
156
+
157
+ usage = {"input_tokens": 0, "output_tokens": 0}
158
+ if hasattr(anyllm_response, "usage") and anyllm_response.usage:
159
+ usage = {
160
+ "input_tokens": getattr(anyllm_response.usage, "prompt_tokens", 0) or 0,
161
+ "output_tokens": getattr(anyllm_response.usage, "completion_tokens", 0) or 0,
162
+ }
163
+
164
+ return {
165
+ "id": msg_id,
166
+ "type": "message",
167
+ "role": "assistant",
168
+ "content": content,
169
+ "model": original_model,
170
+ "stop_reason": stop_reason,
171
+ "stop_sequence": None,
172
+ "usage": usage,
173
+ }
174
+
175
+ async def send_message(
176
+ self,
177
+ body: dict[str, Any],
178
+ headers: dict[str, str],
179
+ ) -> BackendResponse:
180
+ """Send message via any-llm."""
181
+ original_model = body.get("model", "gpt-4o")
182
+
183
+ try:
184
+ messages = self._convert_messages_for_anyllm(body.get("messages", []))
185
+
186
+ kwargs: dict[str, Any] = {
187
+ "model": original_model,
188
+ "provider": self.provider,
189
+ "messages": messages,
190
+ "stream": False,
191
+ }
192
+
193
+ if "max_tokens" in body:
194
+ kwargs["max_tokens"] = body["max_tokens"]
195
+ if "temperature" in body:
196
+ kwargs["temperature"] = body["temperature"]
197
+ if "top_p" in body:
198
+ kwargs["top_p"] = body["top_p"]
199
+ if "stop_sequences" in body:
200
+ kwargs["stop"] = body["stop_sequences"]
201
+
202
+ if "system" in body:
203
+ system = body["system"]
204
+ if isinstance(system, str):
205
+ kwargs["messages"].insert(0, {"role": "system", "content": system})
206
+ elif isinstance(system, list):
207
+ system_text = " ".join(
208
+ s.get("text", "") if isinstance(s, dict) else str(s) for s in system
209
+ )
210
+ kwargs["messages"].insert(0, {"role": "system", "content": system_text})
211
+
212
+ if self.api_key:
213
+ kwargs["api_key"] = self.api_key
214
+ if self.api_base:
215
+ kwargs["api_base"] = self.api_base
216
+
217
+ logger.debug(f"any-llm request: provider={self.provider}, model={original_model}")
218
+
219
+ response = await acompletion(**kwargs)
220
+ anthropic_response = self._to_anthropic_response(response, original_model)
221
+
222
+ return BackendResponse(
223
+ body=anthropic_response,
224
+ status_code=200,
225
+ headers={"content-type": "application/json"},
226
+ )
227
+
228
+ except Exception as e:
229
+ logger.error(f"any-llm error: {e}")
230
+
231
+ error_type = "api_error"
232
+ status_code = 500
233
+
234
+ error_str = str(e).lower()
235
+ if "authentication" in error_str or "api_key" in error_str or "api key" in error_str:
236
+ error_type = "authentication_error"
237
+ status_code = 401
238
+ elif "rate" in error_str or "limit" in error_str:
239
+ error_type = "rate_limit_error"
240
+ status_code = 429
241
+ elif "not found" in error_str or "model" in error_str:
242
+ error_type = "not_found_error"
243
+ status_code = 404
244
+
245
+ return BackendResponse(
246
+ body={
247
+ "type": "error",
248
+ "error": {"type": error_type, "message": str(e)},
249
+ },
250
+ status_code=status_code,
251
+ error=str(e),
252
+ )
253
+
254
+ async def stream_message(
255
+ self,
256
+ body: dict[str, Any],
257
+ headers: dict[str, str],
258
+ ) -> AsyncIterator[StreamEvent]:
259
+ """Stream message via any-llm."""
260
+ original_model = body.get("model", "gpt-4o")
261
+
262
+ try:
263
+ messages = self._convert_messages_for_anyllm(body.get("messages", []))
264
+
265
+ kwargs: dict[str, Any] = {
266
+ "model": original_model,
267
+ "provider": self.provider,
268
+ "messages": messages,
269
+ "stream": True,
270
+ }
271
+
272
+ if "max_tokens" in body:
273
+ kwargs["max_tokens"] = body["max_tokens"]
274
+ if "temperature" in body:
275
+ kwargs["temperature"] = body["temperature"]
276
+ if "system" in body:
277
+ system = body["system"]
278
+ if isinstance(system, str):
279
+ kwargs["messages"].insert(0, {"role": "system", "content": system})
280
+
281
+ if self.api_key:
282
+ kwargs["api_key"] = self.api_key
283
+ if self.api_base:
284
+ kwargs["api_base"] = self.api_base
285
+
286
+ msg_id = f"msg_{uuid.uuid4().hex[:24]}"
287
+
288
+ yield StreamEvent(
289
+ event_type="message_start",
290
+ data={
291
+ "type": "message_start",
292
+ "message": {
293
+ "id": msg_id,
294
+ "type": "message",
295
+ "role": "assistant",
296
+ "content": [],
297
+ "model": original_model,
298
+ "stop_reason": None,
299
+ "stop_sequence": None,
300
+ "usage": {"input_tokens": 0, "output_tokens": 0},
301
+ },
302
+ },
303
+ )
304
+
305
+ yield StreamEvent(
306
+ event_type="content_block_start",
307
+ data={
308
+ "type": "content_block_start",
309
+ "index": 0,
310
+ "content_block": {"type": "text", "text": ""},
311
+ },
312
+ )
313
+
314
+ response = await acompletion(**kwargs)
315
+ output_tokens = 0
316
+
317
+ async for chunk in response:
318
+ if hasattr(chunk, "choices") and chunk.choices:
319
+ delta = chunk.choices[0].delta
320
+ if hasattr(delta, "content") and delta.content:
321
+ yield StreamEvent(
322
+ event_type="content_block_delta",
323
+ data={
324
+ "type": "content_block_delta",
325
+ "index": 0,
326
+ "delta": {"type": "text_delta", "text": delta.content},
327
+ },
328
+ )
329
+ output_tokens += 1
330
+
331
+ yield StreamEvent(
332
+ event_type="content_block_stop",
333
+ data={"type": "content_block_stop", "index": 0},
334
+ )
335
+
336
+ yield StreamEvent(
337
+ event_type="message_delta",
338
+ data={
339
+ "type": "message_delta",
340
+ "delta": {"stop_reason": "end_turn", "stop_sequence": None},
341
+ "usage": {"output_tokens": output_tokens},
342
+ },
343
+ )
344
+
345
+ yield StreamEvent(
346
+ event_type="message_stop",
347
+ data={"type": "message_stop"},
348
+ )
349
+
350
+ except Exception as e:
351
+ logger.error(f"any-llm streaming error: {e}")
352
+ yield StreamEvent(
353
+ event_type="error",
354
+ data={
355
+ "type": "error",
356
+ "error": {"type": "api_error", "message": str(e)},
357
+ },
358
+ )
359
+
360
+ async def send_openai_message(
361
+ self,
362
+ body: dict[str, Any],
363
+ headers: dict[str, str],
364
+ ) -> BackendResponse:
365
+ """Send OpenAI-format message via any-llm (no conversion)."""
366
+ original_model = body.get("model", "gpt-4o")
367
+
368
+ try:
369
+ kwargs: dict[str, Any] = {
370
+ "model": original_model,
371
+ "provider": self.provider,
372
+ "messages": body.get("messages", []),
373
+ "stream": False,
374
+ }
375
+
376
+ for param in [
377
+ "max_tokens",
378
+ "temperature",
379
+ "top_p",
380
+ "stop",
381
+ "tools",
382
+ "tool_choice",
383
+ "response_format",
384
+ "seed",
385
+ "n",
386
+ ]:
387
+ if param in body:
388
+ kwargs[param] = body[param]
389
+
390
+ if self.api_key:
391
+ kwargs["api_key"] = self.api_key
392
+ if self.api_base:
393
+ kwargs["api_base"] = self.api_base
394
+
395
+ logger.debug(f"any-llm OpenAI request: provider={self.provider}, model={original_model}")
396
+
397
+ response = await acompletion(**kwargs)
398
+
399
+ response_dict = {
400
+ "id": response.id,
401
+ "object": "chat.completion",
402
+ "created": response.created,
403
+ "model": original_model,
404
+ "choices": [
405
+ {
406
+ "index": c.index,
407
+ "message": {
408
+ "role": c.message.role,
409
+ "content": c.message.content,
410
+ **(
411
+ {
412
+ "tool_calls": [
413
+ {
414
+ "id": tc.id,
415
+ "type": "function",
416
+ "function": {
417
+ "name": tc.function.name,
418
+ "arguments": tc.function.arguments,
419
+ },
420
+ }
421
+ for tc in c.message.tool_calls
422
+ ]
423
+ }
424
+ if c.message.tool_calls
425
+ else {}
426
+ ),
427
+ },
428
+ "finish_reason": c.finish_reason,
429
+ }
430
+ for c in response.choices
431
+ ],
432
+ "usage": {
433
+ "prompt_tokens": response.usage.prompt_tokens if response.usage else 0,
434
+ "completion_tokens": response.usage.completion_tokens if response.usage else 0,
435
+ "total_tokens": response.usage.total_tokens if response.usage else 0,
436
+ },
437
+ }
438
+
439
+ return BackendResponse(
440
+ body=response_dict,
441
+ status_code=200,
442
+ headers={"content-type": "application/json"},
443
+ )
444
+
445
+ except Exception as e:
446
+ logger.error(f"any-llm OpenAI error: {e}")
447
+
448
+ error_type = "api_error"
449
+ status_code = 500
450
+
451
+ error_str = str(e).lower()
452
+ if "authentication" in error_str or "api_key" in error_str:
453
+ error_type = "invalid_api_key"
454
+ status_code = 401
455
+ elif "rate" in error_str or "limit" in error_str:
456
+ error_type = "rate_limit_exceeded"
457
+ status_code = 429
458
+ elif "not found" in error_str:
459
+ error_type = "model_not_found"
460
+ status_code = 404
461
+
462
+ return BackendResponse(
463
+ body={
464
+ "error": {
465
+ "message": str(e),
466
+ "type": error_type,
467
+ "code": error_type,
468
+ }
469
+ },
470
+ status_code=status_code,
471
+ error=str(e),
472
+ )
473
+
474
+ async def close(self) -> None:
475
+ """Clean up (no-op for any-llm)."""
476
+ pass
headroom/cli/proxy.py CHANGED
@@ -73,9 +73,14 @@ from .main import main
73
  default="anthropic",
74
  help=(
75
  "API backend: 'anthropic' (direct), 'bedrock' (AWS), 'openrouter' (OpenRouter), "
76
- "or 'litellm-<provider>' (e.g., litellm-vertex)"
77
  ),
78
  )
 
 
 
 
 
79
  @click.option(
80
  "--region",
81
  default="us-west-2",
@@ -114,6 +119,7 @@ def proxy(
114
  no_memory_context: bool,
115
  memory_top_k: int,
116
  backend: str,
 
117
  region: str,
118
  bedrock_region: str | None,
119
  bedrock_profile: str | None,
@@ -166,10 +172,11 @@ def proxy(
166
  memory_inject_tools=not no_memory_tools,
167
  memory_inject_context=not no_memory_context,
168
  memory_top_k=memory_top_k,
169
- # Backend (Anthropic direct, Bedrock, or LiteLLM)
170
  backend=backend,
171
  bedrock_region=bedrock_region or region,
172
  bedrock_profile=bedrock_profile,
 
173
  )
174
 
175
  memory_status = "DISABLED"
@@ -180,8 +187,15 @@ def proxy(
180
  backend_status = "Anthropic (direct API)"
181
  backend_section = ""
182
 
183
- if config.backend != "anthropic":
184
- # Get provider config from registry
 
 
 
 
 
 
 
185
  from headroom.backends.litellm import get_provider_config
186
 
187
  provider = config.backend.replace("litellm-", "")
 
73
  default="anthropic",
74
  help=(
75
  "API backend: 'anthropic' (direct), 'bedrock' (AWS), 'openrouter' (OpenRouter), "
76
+ "'anyllm' (any-llm), or 'litellm-<provider>' (e.g., litellm-vertex)"
77
  ),
78
  )
79
+ @click.option(
80
+ "--anyllm-provider",
81
+ default="openai",
82
+ help="Provider for any-llm backend: openai, mistral, groq, ollama, etc. (default: openai)",
83
+ )
84
  @click.option(
85
  "--region",
86
  default="us-west-2",
 
119
  no_memory_context: bool,
120
  memory_top_k: int,
121
  backend: str,
122
+ anyllm_provider: str,
123
  region: str,
124
  bedrock_region: str | None,
125
  bedrock_profile: str | None,
 
172
  memory_inject_tools=not no_memory_tools,
173
  memory_inject_context=not no_memory_context,
174
  memory_top_k=memory_top_k,
175
+ # Backend (Anthropic direct, Bedrock, LiteLLM, or any-llm)
176
  backend=backend,
177
  bedrock_region=bedrock_region or region,
178
  bedrock_profile=bedrock_profile,
179
+ anyllm_provider=anyllm_provider,
180
  )
181
 
182
  memory_status = "DISABLED"
 
187
  backend_status = "Anthropic (direct API)"
188
  backend_section = ""
189
 
190
+ if config.backend == "anyllm" or config.backend.startswith("anyllm-"):
191
+ # any-llm backend
192
+ backend_status = f"{anyllm_provider.title()} via any-llm"
193
+ backend_section = """
194
+ Set credentials for your provider (e.g., OPENAI_API_KEY, MISTRAL_API_KEY)
195
+ Providers: https://mozilla-ai.github.io/any-llm/providers/
196
+ """
197
+ elif config.backend != "anthropic":
198
+ # LiteLLM backend
199
  from headroom.backends.litellm import get_provider_config
200
 
201
  provider = config.backend.replace("litellm-", "")
headroom/proxy/server.py CHANGED
@@ -57,7 +57,7 @@ except ImportError:
57
  sys.path.insert(0, str(Path(__file__).parent.parent.parent))
58
 
59
  from headroom import __version__
60
- from headroom.backends import LiteLLMBackend
61
  from headroom.backends.base import Backend
62
  from headroom.cache.compression_feedback import get_compression_feedback
63
  from headroom.cache.compression_store import get_compression_store
@@ -213,11 +213,13 @@ class ProxyConfig:
213
  openai_api_url: str | None = None # Custom OpenAI API URL override
214
  gemini_api_url: str | None = None # Custom Gemini API URL override
215
 
216
- # Backend: "anthropic" (direct API), "bedrock" (AWS Bedrock), or "litellm-*" (via LiteLLM)
217
  # LiteLLM backends: "litellm-bedrock", "litellm-vertex", "litellm-azure", etc.
 
218
  backend: str = "anthropic"
219
  bedrock_region: str = "us-west-2" # AWS region for Bedrock/LiteLLM
220
  bedrock_profile: str | None = None # AWS profile (optional)
 
221
 
222
  # Optimization
223
  optimize: bool = True
@@ -1080,28 +1082,41 @@ class HeadroomProxy:
1080
  # HTTP client
1081
  self.http_client: httpx.AsyncClient | None = None
1082
 
1083
- # Backend for Anthropic API (direct or via LiteLLM)
1084
- # Supports: "anthropic" (direct), "bedrock", "vertex", or "litellm-<provider>"
1085
  self.anthropic_backend: Backend | None = None
1086
  if config.backend != "anthropic":
1087
- # Normalize backend name: "bedrock" -> "litellm-bedrock"
1088
  backend = config.backend
1089
- if not backend.startswith("litellm-"):
1090
- backend = f"litellm-{backend}"
1091
- provider = backend.replace("litellm-", "")
1092
 
1093
- try:
1094
- self.anthropic_backend = LiteLLMBackend(
1095
- provider=provider,
1096
- region=config.bedrock_region,
1097
- )
1098
- logger.info(
1099
- f"LiteLLM backend enabled (provider={provider}, region={config.bedrock_region})"
1100
- )
1101
- except ImportError as e:
1102
- logger.warning(f"LiteLLM backend not available: {e}")
1103
- except Exception as e:
1104
- logger.error(f"Failed to initialize LiteLLM backend: {e}")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1105
 
1106
  # Request counter for IDs
1107
  self._request_counter = 0
@@ -6527,6 +6542,8 @@ def run_server(
6527
  # Backend status - use provider registry for display info
6528
  if config.backend == "anthropic":
6529
  backend_status = "ANTHROPIC (direct API)"
 
 
6530
  else:
6531
  from headroom.backends.litellm import get_provider_config
6532
 
@@ -6637,12 +6654,12 @@ if __name__ == "__main__":
6637
  "--openai-api-url", help=f"Custom OpenAI API URL (default: {HeadroomProxy.OPENAI_API_URL})"
6638
  )
6639
 
6640
- # Backend (anthropic direct, bedrock, or openrouter)
6641
  parser.add_argument(
6642
  "--backend",
6643
- choices=["anthropic", "bedrock", "openrouter"],
6644
  default="anthropic",
6645
- help="Backend for Anthropic API: 'anthropic' (direct), 'bedrock' (AWS), or 'openrouter' (OpenRouter)",
6646
  )
6647
  parser.add_argument(
6648
  "--bedrock-region",
@@ -6657,6 +6674,11 @@ if __name__ == "__main__":
6657
  "--openrouter-api-key",
6658
  help="OpenRouter API key (or set OPENROUTER_API_KEY env var)",
6659
  )
 
 
 
 
 
6660
 
6661
  # Connection pool (scalability)
6662
  parser.add_argument(
@@ -6791,6 +6813,7 @@ if __name__ == "__main__":
6791
  backend=_get_env_str("HEADROOM_BACKEND", args.backend), # type: ignore[arg-type]
6792
  bedrock_region=_get_env_str("HEADROOM_BEDROCK_REGION", args.bedrock_region),
6793
  bedrock_profile=args.bedrock_profile or os.environ.get("AWS_PROFILE"),
 
6794
  optimize=optimize,
6795
  min_tokens_to_crush=_get_env_int("HEADROOM_MIN_TOKENS", args.min_tokens),
6796
  max_items_after_crush=_get_env_int("HEADROOM_MAX_ITEMS", args.max_items),
 
57
  sys.path.insert(0, str(Path(__file__).parent.parent.parent))
58
 
59
  from headroom import __version__
60
+ from headroom.backends import AnyLLMBackend, LiteLLMBackend
61
  from headroom.backends.base import Backend
62
  from headroom.cache.compression_feedback import get_compression_feedback
63
  from headroom.cache.compression_store import get_compression_store
 
213
  openai_api_url: str | None = None # Custom OpenAI API URL override
214
  gemini_api_url: str | None = None # Custom Gemini API URL override
215
 
216
+ # Backend: "anthropic" (direct API), "litellm-*" (via LiteLLM), or "anyllm" (via any-llm)
217
  # LiteLLM backends: "litellm-bedrock", "litellm-vertex", "litellm-azure", etc.
218
+ # any-llm backends: "anyllm" with --anyllm-provider (openai, mistral, groq, etc.)
219
  backend: str = "anthropic"
220
  bedrock_region: str = "us-west-2" # AWS region for Bedrock/LiteLLM
221
  bedrock_profile: str | None = None # AWS profile (optional)
222
+ anyllm_provider: str = "openai" # any-llm provider (openai, mistral, groq, etc.)
223
 
224
  # Optimization
225
  optimize: bool = True
 
1082
  # HTTP client
1083
  self.http_client: httpx.AsyncClient | None = None
1084
 
1085
+ # Backend for Anthropic API (direct, LiteLLM, or any-llm)
1086
+ # Supports: "anthropic" (direct), "bedrock", "vertex", "litellm-<provider>", or "anyllm"
1087
  self.anthropic_backend: Backend | None = None
1088
  if config.backend != "anthropic":
 
1089
  backend = config.backend
 
 
 
1090
 
1091
+ # Handle any-llm backend
1092
+ if backend == "anyllm" or backend.startswith("anyllm-"):
1093
+ provider = config.anyllm_provider
1094
+ try:
1095
+ self.anthropic_backend = AnyLLMBackend(provider=provider)
1096
+ logger.info(f"any-llm backend enabled (provider={provider})")
1097
+ except ImportError as e:
1098
+ logger.warning(f"any-llm backend not available: {e}")
1099
+ except Exception as e:
1100
+ logger.error(f"Failed to initialize any-llm backend: {e}")
1101
+ else:
1102
+ # Handle LiteLLM backend
1103
+ # Normalize backend name: "bedrock" -> "litellm-bedrock"
1104
+ if not backend.startswith("litellm-"):
1105
+ backend = f"litellm-{backend}"
1106
+ provider = backend.replace("litellm-", "")
1107
+
1108
+ try:
1109
+ self.anthropic_backend = LiteLLMBackend(
1110
+ provider=provider,
1111
+ region=config.bedrock_region,
1112
+ )
1113
+ logger.info(
1114
+ f"LiteLLM backend enabled (provider={provider}, region={config.bedrock_region})"
1115
+ )
1116
+ except ImportError as e:
1117
+ logger.warning(f"LiteLLM backend not available: {e}")
1118
+ except Exception as e:
1119
+ logger.error(f"Failed to initialize LiteLLM backend: {e}")
1120
 
1121
  # Request counter for IDs
1122
  self._request_counter = 0
 
6542
  # Backend status - use provider registry for display info
6543
  if config.backend == "anthropic":
6544
  backend_status = "ANTHROPIC (direct API)"
6545
+ elif config.backend == "anyllm" or config.backend.startswith("anyllm-"):
6546
+ backend_status = f"{config.anyllm_provider.title()} via any-llm"
6547
  else:
6548
  from headroom.backends.litellm import get_provider_config
6549
 
 
6654
  "--openai-api-url", help=f"Custom OpenAI API URL (default: {HeadroomProxy.OPENAI_API_URL})"
6655
  )
6656
 
6657
+ # Backend (anthropic direct, bedrock, openrouter, or anyllm)
6658
  parser.add_argument(
6659
  "--backend",
6660
+ choices=["anthropic", "bedrock", "openrouter", "anyllm"],
6661
  default="anthropic",
6662
+ help="Backend for Anthropic API: 'anthropic' (direct), 'bedrock' (AWS), 'openrouter', or 'anyllm' (any-llm)",
6663
  )
6664
  parser.add_argument(
6665
  "--bedrock-region",
 
6674
  "--openrouter-api-key",
6675
  help="OpenRouter API key (or set OPENROUTER_API_KEY env var)",
6676
  )
6677
+ parser.add_argument(
6678
+ "--anyllm-provider",
6679
+ default="openai",
6680
+ help="any-llm provider: openai, anthropic, mistral, groq, ollama, bedrock, etc. (default: openai)",
6681
+ )
6682
 
6683
  # Connection pool (scalability)
6684
  parser.add_argument(
 
6813
  backend=_get_env_str("HEADROOM_BACKEND", args.backend), # type: ignore[arg-type]
6814
  bedrock_region=_get_env_str("HEADROOM_BEDROCK_REGION", args.bedrock_region),
6815
  bedrock_profile=args.bedrock_profile or os.environ.get("AWS_PROFILE"),
6816
+ anyllm_provider=_get_env_str("HEADROOM_ANYLLM_PROVIDER", args.anyllm_provider),
6817
  optimize=optimize,
6818
  min_tokens_to_crush=_get_env_int("HEADROOM_MIN_TOKENS", args.min_tokens),
6819
  max_items_after_crush=_get_env_int("HEADROOM_MAX_ITEMS", args.max_items),
pyproject.toml CHANGED
@@ -86,6 +86,10 @@ llmlingua = [
86
  code = [
87
  "tree-sitter-language-pack>=0.10.0",
88
  ]
 
 
 
 
89
  # Agno agent framework integration
90
  agno = [
91
  "agno>=1.0.0",
 
86
  code = [
87
  "tree-sitter-language-pack>=0.10.0",
88
  ]
89
+ # any-llm multi-provider backend (requires Python 3.11+)
90
+ anyllm = [
91
+ "any-llm-sdk>=1.0.0",
92
+ ]
93
  # Agno agent framework integration
94
  agno = [
95
  "agno>=1.0.0",