Enhance chat service and local retriever functionality
Browse files- Updated chat_service.py to preserve retrieval results throughout the conversation, ensuring citation-bearing summaries are retained for grounding final outputs.
- Added filtering in stream_chat to prevent structured summary templates from leaking into user-facing responses.
- Modified chroma_rag.py to handle special tokens during encoding, preventing crashes when processing certain content.
- scripts/chat_service.py +18 -0
- scripts/chroma_rag.py +6 -1
scripts/chat_service.py
CHANGED
|
@@ -618,6 +618,13 @@ def build_agent(
|
|
| 618 |
ClearToolUsesEdit(
|
| 619 |
trigger=5_000,
|
| 620 |
keep=3,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 621 |
placeholder="[tool output cleared to save context]",
|
| 622 |
)
|
| 623 |
],
|
|
@@ -953,6 +960,17 @@ async def stream_chat(request: ChatRequest) -> AsyncIterator[ChatEvent]:
|
|
| 953 |
if not isinstance(token, AIMessageChunk):
|
| 954 |
continue
|
| 955 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 956 |
step = str(metadata.get("langgraph_step", ""))
|
| 957 |
if include_reasoning:
|
| 958 |
thought_text = "\n\n".join(
|
|
|
|
| 618 |
ClearToolUsesEdit(
|
| 619 |
trigger=5_000,
|
| 620 |
keep=3,
|
| 621 |
+
# Preserve retrieval results through the whole turn. They
|
| 622 |
+
# are pre-ranked, citation-bearing summaries that the
|
| 623 |
+
# model uses to ground the final synthesis. KB shell
|
| 624 |
+
# outputs (raw file content) get aggressively cleared
|
| 625 |
+
# past the last 3, which is correct because they're
|
| 626 |
+
# large and easy to re-fetch with another `cat`.
|
| 627 |
+
exclude_tools=("retrieve_tutor_context",),
|
| 628 |
placeholder="[tool output cleared to save context]",
|
| 629 |
)
|
| 630 |
],
|
|
|
|
| 960 |
if not isinstance(token, AIMessageChunk):
|
| 961 |
continue
|
| 962 |
|
| 963 |
+
# SummarizationMiddleware fires its own LLM call to compress
|
| 964 |
+
# older history. LangChain tags those calls with
|
| 965 |
+
# `lc_source="summarization"` (see langchain.agents.middleware
|
| 966 |
+
# SummarizationMiddleware._create_summary). Without this
|
| 967 |
+
# filter, the structured summary template ("## SESSION INTENT
|
| 968 |
+
# ...") leaks into the user-facing answer stream. The summary
|
| 969 |
+
# is supposed to be model-internal — like Codex's compact —
|
| 970 |
+
# so we drop those chunks here.
|
| 971 |
+
if metadata.get("lc_source") == "summarization":
|
| 972 |
+
continue
|
| 973 |
+
|
| 974 |
step = str(metadata.get("langgraph_step", ""))
|
| 975 |
if include_reasoning:
|
| 976 |
thought_text = "\n\n".join(
|
scripts/chroma_rag.py
CHANGED
|
@@ -1339,7 +1339,12 @@ class LocalChromaRetriever:
|
|
| 1339 |
if result.score < 0.10:
|
| 1340 |
continue
|
| 1341 |
|
| 1342 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1343 |
if total_tokens + result_tokens > self._token_budget:
|
| 1344 |
break
|
| 1345 |
|
|
|
|
| 1339 |
if result.score < 0.10:
|
| 1340 |
continue
|
| 1341 |
|
| 1342 |
+
# disallowed_special=() so chunks containing literal special-token
|
| 1343 |
+
# strings like "<|endoftext|>" (which sometimes appear verbatim in
|
| 1344 |
+
# tokenizer/training docs) don't crash the encode call.
|
| 1345 |
+
result_tokens = len(
|
| 1346 |
+
self._encoding.encode(result.content, disallowed_special=())
|
| 1347 |
+
)
|
| 1348 |
if total_tokens + result_tokens > self._token_budget:
|
| 1349 |
break
|
| 1350 |
|