omarsol commited on
Commit
4a9c07a
·
1 Parent(s): e73e01f

Enhance chat service and local retriever functionality

Browse files

- Updated chat_service.py to preserve retrieval results throughout the conversation, ensuring citation-bearing summaries are retained for grounding final outputs.
- Added filtering in stream_chat to prevent structured summary templates from leaking into user-facing responses.
- Modified chroma_rag.py to handle special tokens during encoding, preventing crashes when processing certain content.

Files changed (2) hide show
  1. scripts/chat_service.py +18 -0
  2. scripts/chroma_rag.py +6 -1
scripts/chat_service.py CHANGED
@@ -618,6 +618,13 @@ def build_agent(
618
  ClearToolUsesEdit(
619
  trigger=5_000,
620
  keep=3,
 
 
 
 
 
 
 
621
  placeholder="[tool output cleared to save context]",
622
  )
623
  ],
@@ -953,6 +960,17 @@ async def stream_chat(request: ChatRequest) -> AsyncIterator[ChatEvent]:
953
  if not isinstance(token, AIMessageChunk):
954
  continue
955
 
 
 
 
 
 
 
 
 
 
 
 
956
  step = str(metadata.get("langgraph_step", ""))
957
  if include_reasoning:
958
  thought_text = "\n\n".join(
 
618
  ClearToolUsesEdit(
619
  trigger=5_000,
620
  keep=3,
621
+ # Preserve retrieval results through the whole turn. They
622
+ # are pre-ranked, citation-bearing summaries that the
623
+ # model uses to ground the final synthesis. KB shell
624
+ # outputs (raw file content) get aggressively cleared
625
+ # past the last 3, which is correct because they're
626
+ # large and easy to re-fetch with another `cat`.
627
+ exclude_tools=("retrieve_tutor_context",),
628
  placeholder="[tool output cleared to save context]",
629
  )
630
  ],
 
960
  if not isinstance(token, AIMessageChunk):
961
  continue
962
 
963
+ # SummarizationMiddleware fires its own LLM call to compress
964
+ # older history. LangChain tags those calls with
965
+ # `lc_source="summarization"` (see langchain.agents.middleware
966
+ # SummarizationMiddleware._create_summary). Without this
967
+ # filter, the structured summary template ("## SESSION INTENT
968
+ # ...") leaks into the user-facing answer stream. The summary
969
+ # is supposed to be model-internal — like Codex's compact —
970
+ # so we drop those chunks here.
971
+ if metadata.get("lc_source") == "summarization":
972
+ continue
973
+
974
  step = str(metadata.get("langgraph_step", ""))
975
  if include_reasoning:
976
  thought_text = "\n\n".join(
scripts/chroma_rag.py CHANGED
@@ -1339,7 +1339,12 @@ class LocalChromaRetriever:
1339
  if result.score < 0.10:
1340
  continue
1341
 
1342
- result_tokens = len(self._encoding.encode(result.content))
 
 
 
 
 
1343
  if total_tokens + result_tokens > self._token_budget:
1344
  break
1345
 
 
1339
  if result.score < 0.10:
1340
  continue
1341
 
1342
+ # disallowed_special=() so chunks containing literal special-token
1343
+ # strings like "<|endoftext|>" (which sometimes appear verbatim in
1344
+ # tokenizer/training docs) don't crash the encode call.
1345
+ result_tokens = len(
1346
+ self._encoding.encode(result.content, disallowed_special=())
1347
+ )
1348
  if total_tokens + result_tokens > self._token_budget:
1349
  break
1350