fix: preserve DeepSeek thinking events
Browse files- app/chat_service.py +32 -6
- app/deepseek_chat.py +58 -0
- app/provider_events.py +16 -0
- pyproject.toml +1 -0
- tests/test_chat_service.py +115 -1
- tests/test_memory_variants.py +12 -2
- uv.lock +16 -1
app/chat_service.py
CHANGED
|
@@ -41,6 +41,7 @@ from langgraph.store.memory import InMemoryStore
|
|
| 41 |
from langgraph.types import Command
|
| 42 |
|
| 43 |
from .chat_types import ChatEvent, ChatRequest, ChatTurn, SourceMatch
|
|
|
|
| 44 |
from .memory_presets import (
|
| 45 |
DEFAULT_MEMORY_PRESET,
|
| 46 |
MEMORY_PRESETS,
|
|
@@ -78,7 +79,7 @@ from .prompts import build_system_prompt, ensure_kb_agents_instructions
|
|
| 78 |
from .provider_events import (
|
| 79 |
GoogleSearchActivity,
|
| 80 |
extract_anthropic_source_matches,
|
| 81 |
-
|
| 82 |
)
|
| 83 |
from .config import (
|
| 84 |
BM25_INDEX_PATH,
|
|
@@ -906,6 +907,12 @@ def is_anthropic_model(model_name: str) -> bool:
|
|
| 906 |
return provider == "anthropic"
|
| 907 |
|
| 908 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 909 |
def supports_gemini_tool_combination(model_name: str) -> bool:
|
| 910 |
"""True when a Gemini model can mix built-in tools with function tools.
|
| 911 |
|
|
@@ -981,18 +988,23 @@ def _build_chat_model_client(provider_model: str, include_thoughts: bool = False
|
|
| 981 |
max_retries=12,
|
| 982 |
)
|
| 983 |
if provider == "deepseek":
|
| 984 |
-
#
|
|
|
|
|
|
|
| 985 |
# stream_usage=True so the streamed response carries token usage ->
|
| 986 |
# context_stats / cost telemetry populates, including the cached-prefix
|
| 987 |
# tokens that drive the cost comparison (DeepSeek caches prefixes
|
| 988 |
# automatically; cache-hit input is ~50x cheaper than cache-miss). Reads
|
| 989 |
# DEEPSEEK_API_KEY.
|
| 990 |
-
return
|
| 991 |
model=actual_model,
|
| 992 |
-
temperature=1,
|
| 993 |
base_url="https://api.deepseek.com",
|
| 994 |
api_key=os.environ.get("DEEPSEEK_API_KEY"),
|
| 995 |
stream_usage=True,
|
|
|
|
|
|
|
|
|
|
| 996 |
# A few retries ride out transient 429/5xx on long agentic sessions
|
| 997 |
# without failing the whole session (we run evals at concurrency 1).
|
| 998 |
max_retries=6,
|
|
@@ -1284,7 +1296,20 @@ class DeepSeekCacheIsolationMiddleware(AgentMiddleware):
|
|
| 1284 |
if not user_id:
|
| 1285 |
return request
|
| 1286 |
settings = dict(getattr(request, "model_settings", None) or {})
|
| 1287 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1288 |
extra_body["user_id"] = user_id
|
| 1289 |
settings["extra_body"] = extra_body
|
| 1290 |
return request.override(model_settings=settings)
|
|
@@ -2690,6 +2715,7 @@ async def stream_chat(request: ChatRequest) -> AsyncIterator[ChatEvent]:
|
|
| 2690 |
include_reasoning = bool(request.include_reasoning) and (
|
| 2691 |
is_google_genai_model(request.model_name)
|
| 2692 |
or is_anthropic_model(request.model_name)
|
|
|
|
| 2693 |
)
|
| 2694 |
# Gemini streams each thought summary as one complete block; Anthropic
|
| 2695 |
# streams partial fragments of a single thought. The encoder uses this to
|
|
@@ -2777,7 +2803,7 @@ async def stream_chat(request: ChatRequest) -> AsyncIterator[ChatEvent]:
|
|
| 2777 |
|
| 2778 |
step = str(metadata.get("langgraph_node", ""))
|
| 2779 |
if include_reasoning:
|
| 2780 |
-
thought_text = "\n\n".join(
|
| 2781 |
if thought_text:
|
| 2782 |
if first_token_at is None:
|
| 2783 |
first_token_at = time.monotonic()
|
|
|
|
| 41 |
from langgraph.types import Command
|
| 42 |
|
| 43 |
from .chat_types import ChatEvent, ChatRequest, ChatTurn, SourceMatch
|
| 44 |
+
from .deepseek_chat import TutorChatDeepSeek
|
| 45 |
from .memory_presets import (
|
| 46 |
DEFAULT_MEMORY_PRESET,
|
| 47 |
MEMORY_PRESETS,
|
|
|
|
| 79 |
from .provider_events import (
|
| 80 |
GoogleSearchActivity,
|
| 81 |
extract_anthropic_source_matches,
|
| 82 |
+
extract_reasoning_deltas,
|
| 83 |
)
|
| 84 |
from .config import (
|
| 85 |
BM25_INDEX_PATH,
|
|
|
|
| 907 |
return provider == "anthropic"
|
| 908 |
|
| 909 |
|
| 910 |
+
def is_deepseek_model(model_name: str) -> bool:
|
| 911 |
+
provider_model = normalize_model_name(model_name)
|
| 912 |
+
provider, _, _actual_model = provider_model.partition(":")
|
| 913 |
+
return provider == "deepseek"
|
| 914 |
+
|
| 915 |
+
|
| 916 |
def supports_gemini_tool_combination(model_name: str) -> bool:
|
| 917 |
"""True when a Gemini model can mix built-in tools with function tools.
|
| 918 |
|
|
|
|
| 988 |
max_retries=12,
|
| 989 |
)
|
| 990 |
if provider == "deepseek":
|
| 991 |
+
# Use the provider-specific adapter: the generic ChatOpenAI wrapper
|
| 992 |
+
# intentionally discards DeepSeek's non-standard reasoning_content
|
| 993 |
+
# field even though the transport itself is OpenAI-compatible.
|
| 994 |
# stream_usage=True so the streamed response carries token usage ->
|
| 995 |
# context_stats / cost telemetry populates, including the cached-prefix
|
| 996 |
# tokens that drive the cost comparison (DeepSeek caches prefixes
|
| 997 |
# automatically; cache-hit input is ~50x cheaper than cache-miss). Reads
|
| 998 |
# DEEPSEEK_API_KEY.
|
| 999 |
+
return TutorChatDeepSeek(
|
| 1000 |
model=actual_model,
|
| 1001 |
+
temperature=None if include_thoughts else 1,
|
| 1002 |
base_url="https://api.deepseek.com",
|
| 1003 |
api_key=os.environ.get("DEEPSEEK_API_KEY"),
|
| 1004 |
stream_usage=True,
|
| 1005 |
+
extra_body={
|
| 1006 |
+
"thinking": {"type": "enabled" if include_thoughts else "disabled"}
|
| 1007 |
+
},
|
| 1008 |
# A few retries ride out transient 429/5xx on long agentic sessions
|
| 1009 |
# without failing the whole session (we run evals at concurrency 1).
|
| 1010 |
max_retries=6,
|
|
|
|
| 1296 |
if not user_id:
|
| 1297 |
return request
|
| 1298 |
settings = dict(getattr(request, "model_settings", None) or {})
|
| 1299 |
+
model = getattr(request, "model", None)
|
| 1300 |
+
model_extra_body: Any = None
|
| 1301 |
+
# Runnable bindings/fallbacks wrap the underlying provider model. Keep
|
| 1302 |
+
# its DeepSeek thinking toggle when adding the experiment-only user_id;
|
| 1303 |
+
# a call-time extra_body otherwise replaces the model-level mapping.
|
| 1304 |
+
for _ in range(4):
|
| 1305 |
+
model_extra_body = getattr(model, "extra_body", None)
|
| 1306 |
+
if model_extra_body is not None:
|
| 1307 |
+
break
|
| 1308 |
+
model = getattr(model, "runnable", None) or getattr(model, "bound", None)
|
| 1309 |
+
if model is None:
|
| 1310 |
+
break
|
| 1311 |
+
extra_body = dict(model_extra_body or {})
|
| 1312 |
+
extra_body.update(settings.get("extra_body") or {})
|
| 1313 |
extra_body["user_id"] = user_id
|
| 1314 |
settings["extra_body"] = extra_body
|
| 1315 |
return request.override(model_settings=settings)
|
|
|
|
| 2715 |
include_reasoning = bool(request.include_reasoning) and (
|
| 2716 |
is_google_genai_model(request.model_name)
|
| 2717 |
or is_anthropic_model(request.model_name)
|
| 2718 |
+
or is_deepseek_model(request.model_name)
|
| 2719 |
)
|
| 2720 |
# Gemini streams each thought summary as one complete block; Anthropic
|
| 2721 |
# streams partial fragments of a single thought. The encoder uses this to
|
|
|
|
| 2803 |
|
| 2804 |
step = str(metadata.get("langgraph_node", ""))
|
| 2805 |
if include_reasoning:
|
| 2806 |
+
thought_text = "\n\n".join(extract_reasoning_deltas(token))
|
| 2807 |
if thought_text:
|
| 2808 |
if first_token_at is None:
|
| 2809 |
first_token_at = time.monotonic()
|
app/deepseek_chat.py
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""DeepSeek chat-model compatibility helpers.
|
| 2 |
+
|
| 3 |
+
``ChatDeepSeek`` preserves the provider's streamed ``reasoning_content`` in
|
| 4 |
+
LangChain message metadata. DeepSeek thinking-mode tool loops additionally
|
| 5 |
+
require that metadata to be serialized back onto assistant tool-call messages;
|
| 6 |
+
the upstream integration does not currently do that replay step.
|
| 7 |
+
"""
|
| 8 |
+
|
| 9 |
+
from __future__ import annotations
|
| 10 |
+
|
| 11 |
+
from typing import Any
|
| 12 |
+
|
| 13 |
+
from langchain_core.language_models import LanguageModelInput
|
| 14 |
+
from langchain_core.messages import BaseMessage
|
| 15 |
+
from langchain_deepseek import ChatDeepSeek
|
| 16 |
+
|
| 17 |
+
|
| 18 |
+
class TutorChatDeepSeek(ChatDeepSeek):
|
| 19 |
+
"""ChatDeepSeek with reasoning replay for thinking-mode tool calls."""
|
| 20 |
+
|
| 21 |
+
@staticmethod
|
| 22 |
+
def _thinking_enabled(payload: dict[str, Any]) -> bool:
|
| 23 |
+
extra_body = payload.get("extra_body")
|
| 24 |
+
if not isinstance(extra_body, dict):
|
| 25 |
+
return False
|
| 26 |
+
thinking = extra_body.get("thinking")
|
| 27 |
+
return isinstance(thinking, dict) and thinking.get("type") == "enabled"
|
| 28 |
+
|
| 29 |
+
def _original_messages(self, input_: LanguageModelInput) -> list[BaseMessage]:
|
| 30 |
+
return self._convert_input(input_).to_messages()
|
| 31 |
+
|
| 32 |
+
def _get_request_payload(
|
| 33 |
+
self,
|
| 34 |
+
input_: LanguageModelInput,
|
| 35 |
+
*,
|
| 36 |
+
stop: list[str] | None = None,
|
| 37 |
+
**kwargs: Any,
|
| 38 |
+
) -> dict[str, Any]:
|
| 39 |
+
payload = super()._get_request_payload(input_, stop=stop, **kwargs)
|
| 40 |
+
if not self._thinking_enabled(payload):
|
| 41 |
+
return payload
|
| 42 |
+
|
| 43 |
+
original_messages = self._original_messages(input_)
|
| 44 |
+
for index, message in enumerate(payload.get("messages") or []):
|
| 45 |
+
if message.get("role") != "assistant" or not message.get("tool_calls"):
|
| 46 |
+
continue
|
| 47 |
+
reasoning_content = ""
|
| 48 |
+
if index < len(original_messages):
|
| 49 |
+
value = original_messages[index].additional_kwargs.get(
|
| 50 |
+
"reasoning_content"
|
| 51 |
+
)
|
| 52 |
+
if isinstance(value, str):
|
| 53 |
+
reasoning_content = value
|
| 54 |
+
# DeepSeek requires the top-level field to exist on every replayed
|
| 55 |
+
# assistant tool-call message in thinking mode. An empty value is
|
| 56 |
+
# still preferable to omitting the field and receiving a 400.
|
| 57 |
+
message["reasoning_content"] = reasoning_content
|
| 58 |
+
return payload
|
app/provider_events.py
CHANGED
|
@@ -40,6 +40,22 @@ def extract_thought_summaries(content: Any) -> list[str]:
|
|
| 40 |
return thoughts
|
| 41 |
|
| 42 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 43 |
def extract_web_search_queries(response_metadata: Any) -> list[str]:
|
| 44 |
"""Pull the queries Gemini ran against google_search from grounding metadata."""
|
| 45 |
if not isinstance(response_metadata, dict):
|
|
|
|
| 40 |
return thoughts
|
| 41 |
|
| 42 |
|
| 43 |
+
def extract_reasoning_deltas(message: Any) -> list[str]:
|
| 44 |
+
"""Extract provider reasoning fragments from a streamed LangChain message.
|
| 45 |
+
|
| 46 |
+
Gemini and Anthropic expose reasoning as typed content blocks. DeepSeek's
|
| 47 |
+
provider adapter preserves its sibling ``reasoning_content`` response field
|
| 48 |
+
in ``additional_kwargs`` instead.
|
| 49 |
+
"""
|
| 50 |
+
thoughts = extract_thought_summaries(getattr(message, "content", None))
|
| 51 |
+
additional_kwargs = getattr(message, "additional_kwargs", None)
|
| 52 |
+
if isinstance(additional_kwargs, dict):
|
| 53 |
+
reasoning_content = additional_kwargs.get("reasoning_content")
|
| 54 |
+
if isinstance(reasoning_content, str) and reasoning_content.strip():
|
| 55 |
+
thoughts.append(reasoning_content)
|
| 56 |
+
return thoughts
|
| 57 |
+
|
| 58 |
+
|
| 59 |
def extract_web_search_queries(response_metadata: Any) -> list[str]:
|
| 60 |
"""Pull the queries Gemini ran against google_search from grounding metadata."""
|
| 61 |
if not isinstance(response_metadata, dict):
|
pyproject.toml
CHANGED
|
@@ -14,6 +14,7 @@ dependencies = [
|
|
| 14 |
"huggingface-hub",
|
| 15 |
"langchain",
|
| 16 |
"langchain-anthropic",
|
|
|
|
| 17 |
"langchain-google-genai",
|
| 18 |
"langchain-openai",
|
| 19 |
"llama-index",
|
|
|
|
| 14 |
"huggingface-hub",
|
| 15 |
"langchain",
|
| 16 |
"langchain-anthropic",
|
| 17 |
+
"langchain-deepseek>=1.1.0",
|
| 18 |
"langchain-google-genai",
|
| 19 |
"langchain-openai",
|
| 20 |
"llama-index",
|
tests/test_chat_service.py
CHANGED
|
@@ -10,6 +10,7 @@ from langchain_core.messages import AIMessage, AIMessageChunk, HumanMessage, Too
|
|
| 10 |
from langchain_core.runnables.fallbacks import RunnableWithFallbacks
|
| 11 |
|
| 12 |
from app.config import DEEPSEEK_DIRECT_MODEL_NAME, GEMINI_FALLBACK_MODEL_NAME
|
|
|
|
| 13 |
from app.chat_service import (
|
| 14 |
THREAD_IDLE_TTL_SECONDS,
|
| 15 |
_claim_kb_command_budget,
|
|
@@ -130,6 +131,32 @@ class FakeAnswerAgent(FakeAgent):
|
|
| 130 |
}
|
| 131 |
|
| 132 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 133 |
class FakeToolThenTextAgent(FakeAgent):
|
| 134 |
"""Streams a tool-call token first, then streamed answer text, so a turn
|
| 135 |
sets first_token_at (the tool call) strictly before first_text_at (the
|
|
@@ -421,9 +448,12 @@ class ChatServiceTestCase(unittest.TestCase):
|
|
| 421 |
model = build_chat_model(DEEPSEEK_DIRECT_MODEL_NAME)
|
| 422 |
|
| 423 |
self.assertIsInstance(model, RunnableWithFallbacks)
|
|
|
|
| 424 |
self.assertEqual(model.runnable.model_name, "deepseek-v4-flash")
|
|
|
|
| 425 |
self.assertEqual(
|
| 426 |
-
|
|
|
|
| 427 |
)
|
| 428 |
self.assertEqual(len(model.fallbacks), 1)
|
| 429 |
self.assertEqual(
|
|
@@ -442,8 +472,58 @@ class ChatServiceTestCase(unittest.TestCase):
|
|
| 442 |
model = build_chat_model(DEEPSEEK_DIRECT_MODEL_NAME)
|
| 443 |
|
| 444 |
self.assertNotIsInstance(model, RunnableWithFallbacks)
|
|
|
|
| 445 |
self.assertEqual(model.model_name, "deepseek-v4-flash")
|
| 446 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 447 |
def test_deepseek_direct_default_uses_gemini_when_deepseek_key_missing(
|
| 448 |
self,
|
| 449 |
) -> None:
|
|
@@ -757,6 +837,40 @@ class ChatServiceTestCase(unittest.TestCase):
|
|
| 757 |
# latency never exceeds time-to-first-answer.
|
| 758 |
self.assertLessEqual(first_token, ttft)
|
| 759 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 760 |
def test_stream_chat_resolves_shell_citation_after_final_answer(self) -> None:
|
| 761 |
agent = FakeStreamingAgent([])
|
| 762 |
self.addCleanup(_drop_thread_record, "thread_rg")
|
|
|
|
| 10 |
from langchain_core.runnables.fallbacks import RunnableWithFallbacks
|
| 11 |
|
| 12 |
from app.config import DEEPSEEK_DIRECT_MODEL_NAME, GEMINI_FALLBACK_MODEL_NAME
|
| 13 |
+
from app.deepseek_chat import TutorChatDeepSeek
|
| 14 |
from app.chat_service import (
|
| 15 |
THREAD_IDLE_TTL_SECONDS,
|
| 16 |
_claim_kb_command_budget,
|
|
|
|
| 131 |
}
|
| 132 |
|
| 133 |
|
| 134 |
+
class FakeDeepSeekReasoningAgent(FakeAgent):
|
| 135 |
+
async def astream(self, *args, **kwargs):
|
| 136 |
+
for fragment in ("Inspect ", "the course sources."):
|
| 137 |
+
yield {
|
| 138 |
+
"type": "messages",
|
| 139 |
+
"data": (
|
| 140 |
+
AIMessageChunk(
|
| 141 |
+
content="",
|
| 142 |
+
additional_kwargs={"reasoning_content": fragment},
|
| 143 |
+
),
|
| 144 |
+
{"langgraph_node": "model"},
|
| 145 |
+
),
|
| 146 |
+
}
|
| 147 |
+
yield {
|
| 148 |
+
"type": "messages",
|
| 149 |
+
"data": (
|
| 150 |
+
AIMessageChunk(content="Final answer"),
|
| 151 |
+
{"langgraph_node": "model"},
|
| 152 |
+
),
|
| 153 |
+
}
|
| 154 |
+
yield {
|
| 155 |
+
"type": "updates",
|
| 156 |
+
"data": {"model": {"messages": [AIMessage(content="Final answer")]}},
|
| 157 |
+
}
|
| 158 |
+
|
| 159 |
+
|
| 160 |
class FakeToolThenTextAgent(FakeAgent):
|
| 161 |
"""Streams a tool-call token first, then streamed answer text, so a turn
|
| 162 |
sets first_token_at (the tool call) strictly before first_text_at (the
|
|
|
|
| 448 |
model = build_chat_model(DEEPSEEK_DIRECT_MODEL_NAME)
|
| 449 |
|
| 450 |
self.assertIsInstance(model, RunnableWithFallbacks)
|
| 451 |
+
self.assertIsInstance(model.runnable, TutorChatDeepSeek)
|
| 452 |
self.assertEqual(model.runnable.model_name, "deepseek-v4-flash")
|
| 453 |
+
self.assertEqual(str(model.runnable.api_base), "https://api.deepseek.com")
|
| 454 |
self.assertEqual(
|
| 455 |
+
model.runnable.extra_body,
|
| 456 |
+
{"thinking": {"type": "disabled"}},
|
| 457 |
)
|
| 458 |
self.assertEqual(len(model.fallbacks), 1)
|
| 459 |
self.assertEqual(
|
|
|
|
| 472 |
model = build_chat_model(DEEPSEEK_DIRECT_MODEL_NAME)
|
| 473 |
|
| 474 |
self.assertNotIsInstance(model, RunnableWithFallbacks)
|
| 475 |
+
self.assertIsInstance(model, TutorChatDeepSeek)
|
| 476 |
self.assertEqual(model.model_name, "deepseek-v4-flash")
|
| 477 |
|
| 478 |
+
def test_deepseek_reasoning_mode_is_explicitly_enabled(self) -> None:
|
| 479 |
+
with patch.dict(
|
| 480 |
+
os.environ,
|
| 481 |
+
{"DEEPSEEK_API_KEY": "deepseek-test-key"},
|
| 482 |
+
clear=True,
|
| 483 |
+
):
|
| 484 |
+
model = build_chat_model(
|
| 485 |
+
DEEPSEEK_DIRECT_MODEL_NAME,
|
| 486 |
+
include_thoughts=True,
|
| 487 |
+
)
|
| 488 |
+
|
| 489 |
+
self.assertIsInstance(model, TutorChatDeepSeek)
|
| 490 |
+
self.assertIsNone(model.temperature)
|
| 491 |
+
self.assertEqual(
|
| 492 |
+
model.extra_body,
|
| 493 |
+
{"thinking": {"type": "enabled"}},
|
| 494 |
+
)
|
| 495 |
+
|
| 496 |
+
def test_deepseek_thinking_tool_call_replays_reasoning_content(self) -> None:
|
| 497 |
+
model = TutorChatDeepSeek(
|
| 498 |
+
model="deepseek-v4-flash",
|
| 499 |
+
api_key="deepseek-test-key",
|
| 500 |
+
extra_body={"thinking": {"type": "enabled"}},
|
| 501 |
+
)
|
| 502 |
+
messages = [
|
| 503 |
+
HumanMessage(content="Find the course repository"),
|
| 504 |
+
AIMessage(
|
| 505 |
+
content="",
|
| 506 |
+
tool_calls=[
|
| 507 |
+
{
|
| 508 |
+
"id": "call_1",
|
| 509 |
+
"name": "retrieve_tutor_context",
|
| 510 |
+
"args": {"query": "course repository"},
|
| 511 |
+
}
|
| 512 |
+
],
|
| 513 |
+
additional_kwargs={
|
| 514 |
+
"reasoning_content": "I should search the selected course."
|
| 515 |
+
},
|
| 516 |
+
),
|
| 517 |
+
ToolMessage(content="result", tool_call_id="call_1"),
|
| 518 |
+
]
|
| 519 |
+
|
| 520 |
+
payload = model._get_request_payload(messages)
|
| 521 |
+
|
| 522 |
+
self.assertEqual(
|
| 523 |
+
payload["messages"][1]["reasoning_content"],
|
| 524 |
+
"I should search the selected course.",
|
| 525 |
+
)
|
| 526 |
+
|
| 527 |
def test_deepseek_direct_default_uses_gemini_when_deepseek_key_missing(
|
| 528 |
self,
|
| 529 |
) -> None:
|
|
|
|
| 837 |
# latency never exceeds time-to-first-answer.
|
| 838 |
self.assertLessEqual(first_token, ttft)
|
| 839 |
|
| 840 |
+
def test_stream_chat_emits_deepseek_reasoning_content(self) -> None:
|
| 841 |
+
agent = FakeDeepSeekReasoningAgent([])
|
| 842 |
+
build_agent_mock = MagicMock(return_value=agent)
|
| 843 |
+
self.addCleanup(_drop_thread_record, "thread_deepseek_reasoning")
|
| 844 |
+
request = ChatRequest(
|
| 845 |
+
query="Find the course repository",
|
| 846 |
+
source_keys=("peft",),
|
| 847 |
+
model_name=DEEPSEEK_DIRECT_MODEL_NAME,
|
| 848 |
+
include_reasoning=True,
|
| 849 |
+
enabled_tools=(),
|
| 850 |
+
)
|
| 851 |
+
|
| 852 |
+
async def collect_events():
|
| 853 |
+
return [event async for event in stream_chat(request)]
|
| 854 |
+
|
| 855 |
+
with (
|
| 856 |
+
patch("app.chat_service.build_agent", build_agent_mock),
|
| 857 |
+
patch(
|
| 858 |
+
"app.chat_service.new_thread_id",
|
| 859 |
+
return_value="thread_deepseek_reasoning",
|
| 860 |
+
),
|
| 861 |
+
):
|
| 862 |
+
events = asyncio.run(collect_events())
|
| 863 |
+
|
| 864 |
+
reasoning = [
|
| 865 |
+
event.data["text"] for event in events if event.type == "reasoning_delta"
|
| 866 |
+
]
|
| 867 |
+
self.assertEqual(reasoning, ["Inspect ", "the course sources."])
|
| 868 |
+
self.assertEqual(
|
| 869 |
+
[event.data["text"] for event in events if event.type == "text_delta"],
|
| 870 |
+
["Final answer"],
|
| 871 |
+
)
|
| 872 |
+
self.assertTrue(build_agent_mock.call_args.kwargs["include_thoughts"])
|
| 873 |
+
|
| 874 |
def test_stream_chat_resolves_shell_citation_after_final_answer(self) -> None:
|
| 875 |
agent = FakeStreamingAgent([])
|
| 876 |
self.addCleanup(_drop_thread_record, "thread_rg")
|
tests/test_memory_variants.py
CHANGED
|
@@ -383,14 +383,24 @@ class ExperimentCompactionMiddlewareTests(unittest.TestCase):
|
|
| 383 |
context=SimpleNamespace(cache_user_id="eval_abc", kb_session_id="turn")
|
| 384 |
)
|
| 385 |
request = SimpleNamespace(
|
| 386 |
-
runtime=runtime,
|
|
|
|
|
|
|
|
|
|
|
|
|
| 387 |
)
|
| 388 |
request.override = lambda **updates: SimpleNamespace(
|
| 389 |
runtime=runtime,
|
| 390 |
model_settings=updates.get("model_settings", request.model_settings),
|
| 391 |
)
|
| 392 |
isolated = DeepSeekCacheIsolationMiddleware()._isolate(request)
|
| 393 |
-
self.assertEqual(
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 394 |
|
| 395 |
def test_agent_request_guard_fails_before_model_handler(self) -> None:
|
| 396 |
runtime = SimpleNamespace(
|
|
|
|
| 383 |
context=SimpleNamespace(cache_user_id="eval_abc", kb_session_id="turn")
|
| 384 |
)
|
| 385 |
request = SimpleNamespace(
|
| 386 |
+
runtime=runtime,
|
| 387 |
+
model=SimpleNamespace(extra_body={"thinking": {"type": "enabled"}}),
|
| 388 |
+
model_settings={},
|
| 389 |
+
messages=[],
|
| 390 |
+
system_message=None,
|
| 391 |
)
|
| 392 |
request.override = lambda **updates: SimpleNamespace(
|
| 393 |
runtime=runtime,
|
| 394 |
model_settings=updates.get("model_settings", request.model_settings),
|
| 395 |
)
|
| 396 |
isolated = DeepSeekCacheIsolationMiddleware()._isolate(request)
|
| 397 |
+
self.assertEqual(
|
| 398 |
+
isolated.model_settings["extra_body"],
|
| 399 |
+
{
|
| 400 |
+
"thinking": {"type": "enabled"},
|
| 401 |
+
"user_id": "eval_abc",
|
| 402 |
+
},
|
| 403 |
+
)
|
| 404 |
|
| 405 |
def test_agent_request_guard_fails_before_model_handler(self) -> None:
|
| 406 |
runtime = SimpleNamespace(
|
uv.lock
CHANGED
|
@@ -22,6 +22,7 @@ dependencies = [
|
|
| 22 |
{ name = "huggingface-hub" },
|
| 23 |
{ name = "langchain" },
|
| 24 |
{ name = "langchain-anthropic" },
|
|
|
|
| 25 |
{ name = "langchain-google-genai" },
|
| 26 |
{ name = "langchain-openai" },
|
| 27 |
{ name = "llama-index" },
|
|
@@ -61,6 +62,7 @@ requires-dist = [
|
|
| 61 |
{ name = "huggingface-hub" },
|
| 62 |
{ name = "langchain" },
|
| 63 |
{ name = "langchain-anthropic" },
|
|
|
|
| 64 |
{ name = "langchain-google-genai" },
|
| 65 |
{ name = "langchain-openai" },
|
| 66 |
{ name = "llama-index" },
|
|
@@ -2016,6 +2018,19 @@ wheels = [
|
|
| 2016 |
{ url = "https://files.pythonhosted.org/packages/13/d6/bdf6f0481cc57ef300d6b1eb48cf1400c0409be715d6eb3cabadd1142a09/langchain_core-1.4.8-py3-none-any.whl", hash = "sha256:d84c28b05e3ba8d4271d0827aad5b592ccdaaf986e76768c23503f0a2045e8aa", size = 557416, upload-time = "2026-06-18T19:39:21.902Z" },
|
| 2017 |
]
|
| 2018 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 2019 |
[[package]]
|
| 2020 |
name = "langchain-google-genai"
|
| 2021 |
version = "4.2.7"
|
|
@@ -3135,7 +3150,7 @@ name = "pexpect"
|
|
| 3135 |
version = "4.9.0"
|
| 3136 |
source = { registry = "https://pypi.org/simple" }
|
| 3137 |
dependencies = [
|
| 3138 |
-
{ name = "ptyprocess" },
|
| 3139 |
]
|
| 3140 |
sdist = { url = "https://files.pythonhosted.org/packages/42/92/cc564bf6381ff43ce1f4d06852fc19a2f11d180f23dc32d9588bee2f149d/pexpect-4.9.0.tar.gz", hash = "sha256:ee7d41123f3c9911050ea2c2dac107568dc43b2d3b0c7557a33212c398ead30f", size = 166450, upload-time = "2023-11-25T09:07:26.339Z" }
|
| 3141 |
wheels = [
|
|
|
|
| 22 |
{ name = "huggingface-hub" },
|
| 23 |
{ name = "langchain" },
|
| 24 |
{ name = "langchain-anthropic" },
|
| 25 |
+
{ name = "langchain-deepseek" },
|
| 26 |
{ name = "langchain-google-genai" },
|
| 27 |
{ name = "langchain-openai" },
|
| 28 |
{ name = "llama-index" },
|
|
|
|
| 62 |
{ name = "huggingface-hub" },
|
| 63 |
{ name = "langchain" },
|
| 64 |
{ name = "langchain-anthropic" },
|
| 65 |
+
{ name = "langchain-deepseek", specifier = ">=1.1.0" },
|
| 66 |
{ name = "langchain-google-genai" },
|
| 67 |
{ name = "langchain-openai" },
|
| 68 |
{ name = "llama-index" },
|
|
|
|
| 2018 |
{ url = "https://files.pythonhosted.org/packages/13/d6/bdf6f0481cc57ef300d6b1eb48cf1400c0409be715d6eb3cabadd1142a09/langchain_core-1.4.8-py3-none-any.whl", hash = "sha256:d84c28b05e3ba8d4271d0827aad5b592ccdaaf986e76768c23503f0a2045e8aa", size = 557416, upload-time = "2026-06-18T19:39:21.902Z" },
|
| 2019 |
]
|
| 2020 |
|
| 2021 |
+
[[package]]
|
| 2022 |
+
name = "langchain-deepseek"
|
| 2023 |
+
version = "1.1.0"
|
| 2024 |
+
source = { registry = "https://pypi.org/simple" }
|
| 2025 |
+
dependencies = [
|
| 2026 |
+
{ name = "langchain-core" },
|
| 2027 |
+
{ name = "langchain-openai" },
|
| 2028 |
+
]
|
| 2029 |
+
sdist = { url = "https://files.pythonhosted.org/packages/15/f4/200792013d86406aa02aba8d1b536f8877a832372702189aef0c8c7a8329/langchain_deepseek-1.1.0.tar.gz", hash = "sha256:5ec8128cabae3d71366e2f297dea2153649be317a4be7f135cfcad1160523bf6", size = 138078, upload-time = "2026-06-03T19:07:11.887Z" }
|
| 2030 |
+
wheels = [
|
| 2031 |
+
{ url = "https://files.pythonhosted.org/packages/8c/1a/d18b7f985c35c635503a6cd426161933ffc7785f228d92447c63b026299a/langchain_deepseek-1.1.0-py3-none-any.whl", hash = "sha256:14813cb413a97a5cce95118da253cfd64dce50537b7381b7c5d0ecf11d2a7032", size = 10061, upload-time = "2026-06-03T19:07:11.04Z" },
|
| 2032 |
+
]
|
| 2033 |
+
|
| 2034 |
[[package]]
|
| 2035 |
name = "langchain-google-genai"
|
| 2036 |
version = "4.2.7"
|
|
|
|
| 3150 |
version = "4.9.0"
|
| 3151 |
source = { registry = "https://pypi.org/simple" }
|
| 3152 |
dependencies = [
|
| 3153 |
+
{ name = "ptyprocess", marker = "sys_platform != 'win32'" },
|
| 3154 |
]
|
| 3155 |
sdist = { url = "https://files.pythonhosted.org/packages/42/92/cc564bf6381ff43ce1f4d06852fc19a2f11d180f23dc32d9588bee2f149d/pexpect-4.9.0.tar.gz", hash = "sha256:ee7d41123f3c9911050ea2c2dac107568dc43b2d3b0c7557a33212c398ead30f", size = 166450, upload-time = "2023-11-25T09:07:26.339Z" }
|
| 3156 |
wheels = [
|