Download utils/chunking.py from trungnd7112004/FastAPI-backend-chatbotRAG: direct link, hf CLI and curl.
- Browser
- Download file 522 Bytes
-
https://huggingface.co/spaces/trungnd7112004/FastAPI-backend-chatbotRAG/resolve/f367ea8e626a14cd88e6a27ae2e6513bad8914b6/utils/chunking.py
- Command line
-
hf download hf://spaces/trungnd7112004/FastAPI-backend-chatbotRAG@f367ea8e626a14cd88e6a27ae2e6513bad8914b6/utils/chunking.py
-
curl -L -o chunking.py https://huggingface.co/spaces/trungnd7112004/FastAPI-backend-chatbotRAG/resolve/f367ea8e626a14cd88e6a27ae2e6513bad8914b6/utils/chunking.py
522 Bytes
| from langchain.text_splitter import MarkdownHeaderTextSplitter | |
| from langchain.schema import Document | |
| def split_text_by_markdown(input_md: str) -> list: | |
| headers_to_split_on = [ | |
| ("#", "Header 1"), | |
| ("##", "Header 2"), | |
| ("###", "Header 3"), | |
| ] | |
| splitter = MarkdownHeaderTextSplitter(headers_to_split_on=headers_to_split_on) | |
| chunks = splitter.split_text(input_md) | |
| documents = [Document(page_content=chunk.page_content, metadata=chunk.metadata) for chunk in chunks] | |
| return documents |