Merge feature/hf-deployment into master
Browse files- .github/workflows/push-to-hf-space.yml +37 -0
- .gitignore +0 -7
- Dockerfile +29 -0
- guards/input.py +30 -0
- guards/output.py +27 -0
- rag/build_vector_store.py +40 -0
- rag/retriever.py +21 -0
- requirements.txt +3 -4
- startup.sh +4 -0
.github/workflows/push-to-hf-space.yml
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
name: Push to Hugging Face Space
|
| 2 |
+
|
| 3 |
+
on:
|
| 4 |
+
push:
|
| 5 |
+
branches: [ feature/hf-deployment ] # change if you use another default branch
|
| 6 |
+
|
| 7 |
+
jobs:
|
| 8 |
+
sync:
|
| 9 |
+
runs-on: ubuntu-latest
|
| 10 |
+
|
| 11 |
+
steps:
|
| 12 |
+
- name: Check out repo
|
| 13 |
+
uses: actions/checkout@v4
|
| 14 |
+
with:
|
| 15 |
+
fetch-depth: 0 # keep history; helpful for subtree/sparse etc.
|
| 16 |
+
|
| 17 |
+
- name: Set up Git LFS (if you track large files)
|
| 18 |
+
run: |
|
| 19 |
+
sudo apt-get update
|
| 20 |
+
sudo apt-get install -y git-lfs
|
| 21 |
+
git lfs install
|
| 22 |
+
|
| 23 |
+
- name: Push to Space
|
| 24 |
+
env:
|
| 25 |
+
HF_TOKEN: ${{ secrets.HF_TOKEN }}
|
| 26 |
+
HF_SPACE: pxdelta/RAG # or use ${{ secrets.HF_SPACE }}
|
| 27 |
+
run: |
|
| 28 |
+
# Configure git identity for the CI push
|
| 29 |
+
git config user.name "github-actions[bot]"
|
| 30 |
+
git config user.email "github-actions[bot]@users.noreply.github.com"
|
| 31 |
+
|
| 32 |
+
# Add the Space as an authenticated remote.
|
| 33 |
+
# Username can be anything (often 'oauth2' or your HF username); token goes in the password slot.
|
| 34 |
+
git remote add hf "https://oauth2:${HF_TOKEN}@huggingface.co/spaces/${HF_SPACE}"
|
| 35 |
+
|
| 36 |
+
# Push the current GitHub branch (e.g., main) to the Space's main branch
|
| 37 |
+
git push hf HEAD:main --force
|
.gitignore
CHANGED
|
@@ -2,10 +2,6 @@
|
|
| 2 |
.env
|
| 3 |
venv/
|
| 4 |
env/
|
| 5 |
-
.venv
|
| 6 |
-
|
| 7 |
-
notes.txt
|
| 8 |
-
test.ipynb
|
| 9 |
|
| 10 |
# Python
|
| 11 |
__pycache__/
|
|
@@ -22,6 +18,3 @@ Thumbs.db
|
|
| 22 |
# Generated files
|
| 23 |
database/university.db
|
| 24 |
rag/vector_store/
|
| 25 |
-
|
| 26 |
-
|
| 27 |
-
secrets_local.py
|
|
|
|
| 2 |
.env
|
| 3 |
venv/
|
| 4 |
env/
|
|
|
|
|
|
|
|
|
|
|
|
|
| 5 |
|
| 6 |
# Python
|
| 7 |
__pycache__/
|
|
|
|
| 18 |
# Generated files
|
| 19 |
database/university.db
|
| 20 |
rag/vector_store/
|
|
|
|
|
|
|
|
|
Dockerfile
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Use the official Python image
|
| 2 |
+
FROM python:3.10-slim
|
| 3 |
+
|
| 4 |
+
# Create a non-root user
|
| 5 |
+
RUN useradd -m -u 1000 user
|
| 6 |
+
|
| 7 |
+
# Set the working directory
|
| 8 |
+
WORKDIR /app
|
| 9 |
+
|
| 10 |
+
# Copy all application files
|
| 11 |
+
COPY . .
|
| 12 |
+
|
| 13 |
+
# Grant ownership of the app directory to the user
|
| 14 |
+
RUN chown -R user:user /app
|
| 15 |
+
|
| 16 |
+
# Switch to the non-root user
|
| 17 |
+
USER user
|
| 18 |
+
|
| 19 |
+
# Install dependencies
|
| 20 |
+
RUN pip install --user -r requirements.txt
|
| 21 |
+
|
| 22 |
+
# Make the startup script executable
|
| 23 |
+
RUN chmod +x startup.sh
|
| 24 |
+
|
| 25 |
+
# Add the user's local bin directory to the PATH
|
| 26 |
+
ENV PATH="/home/user/.local/bin:${PATH}"
|
| 27 |
+
|
| 28 |
+
# Set the entrypoint to the startup script
|
| 29 |
+
CMD ["./startup.sh"]
|
guards/input.py
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import re
|
| 2 |
+
|
| 3 |
+
def is_valid(query: str) -> bool:
|
| 4 |
+
"""
|
| 5 |
+
Validates the user's query.
|
| 6 |
+
"""
|
| 7 |
+
# Check for query length
|
| 8 |
+
if len(query) > 500:
|
| 9 |
+
return False
|
| 10 |
+
|
| 11 |
+
# Check for SQL injection patterns
|
| 12 |
+
sql_injection_patterns = [
|
| 13 |
+
r"(\s*(--|#|;))",
|
| 14 |
+
r"(\s*(union|select|insert|update|delete|drop|alter)\s+)",
|
| 15 |
+
]
|
| 16 |
+
for pattern in sql_injection_patterns:
|
| 17 |
+
if re.search(pattern, query, re.IGNORECASE):
|
| 18 |
+
return False
|
| 19 |
+
|
| 20 |
+
return True
|
| 21 |
+
|
| 22 |
+
if __name__ == '__main__':
|
| 23 |
+
# Example usage
|
| 24 |
+
valid_query = "What are the computer science courses?"
|
| 25 |
+
invalid_query_long = "a" * 501
|
| 26 |
+
invalid_query_sql = "SELECT * FROM students;"
|
| 27 |
+
|
| 28 |
+
print(f"'{valid_query}' is valid: {is_valid(valid_query)}")
|
| 29 |
+
print(f"'long query' is valid: {is_valid(invalid_query_long)}")
|
| 30 |
+
print(f"'{invalid_query_sql}' is valid: {is_valid(invalid_query_sql)}")
|
guards/output.py
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import re
|
| 2 |
+
|
| 3 |
+
def sanitize(response: str) -> str:
|
| 4 |
+
"""
|
| 5 |
+
Sanitizes the LLM's response.
|
| 6 |
+
"""
|
| 7 |
+
# Redact email addresses
|
| 8 |
+
response = re.sub(r"[a-zA-Z0-9_.+-]+@[a-zA-Z0-9-]+\.[a-zA-Z0-9-.]+", "[REDACTED EMAIL]", response)
|
| 9 |
+
|
| 10 |
+
# Check for hallucination patterns
|
| 11 |
+
hallucination_patterns = [
|
| 12 |
+
r"as an ai language model",
|
| 13 |
+
r"i cannot answer that question",
|
| 14 |
+
]
|
| 15 |
+
for pattern in hallucination_patterns:
|
| 16 |
+
if re.search(pattern, response, re.IGNORECASE):
|
| 17 |
+
return "I'm sorry, I don't have enough information to answer that question."
|
| 18 |
+
|
| 19 |
+
return response
|
| 20 |
+
|
| 21 |
+
if __name__ == '__main__':
|
| 22 |
+
# Example usage
|
| 23 |
+
valid_response = "Professor Smith's email is john.smith@university.edu."
|
| 24 |
+
hallucination_response = "As an AI language model, I cannot provide that information."
|
| 25 |
+
|
| 26 |
+
print(f"Original: '{valid_response}'\nSanitized: '{sanitize(valid_response)}'")
|
| 27 |
+
print(f"\nOriginal: '{hallucination_response}'\nSanitized: '{sanitize(hallucination_response)}'")
|
rag/build_vector_store.py
CHANGED
|
@@ -1,6 +1,7 @@
|
|
| 1 |
import sqlite3
|
| 2 |
import chromadb
|
| 3 |
from sentence_transformers import SentenceTransformer
|
|
|
|
| 4 |
from pathlib import Path
|
| 5 |
import sys
|
| 6 |
sys.path.append(str(Path(__file__).parent.parent))
|
|
@@ -10,10 +11,17 @@ def build_vector_store():
|
|
| 10 |
"""
|
| 11 |
Builds a persistent vector store from the data in the SQLite database,
|
| 12 |
embedding information about students, faculty, and courses.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 13 |
"""
|
| 14 |
conn = sqlite3.connect('database/university.db')
|
| 15 |
cursor = conn.cursor()
|
| 16 |
|
|
|
|
| 17 |
documents = []
|
| 18 |
print("Creating student docs")
|
| 19 |
# === Build Student Documents ===
|
|
@@ -89,6 +97,35 @@ def build_vector_store():
|
|
| 89 |
client = chromadb.PersistentClient(path="rag/vector_store")
|
| 90 |
collection = client.get_or_create_collection("university_data")
|
| 91 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 92 |
collection.add(
|
| 93 |
embeddings=embeddings,
|
| 94 |
documents=documents,
|
|
@@ -97,6 +134,9 @@ def build_vector_store():
|
|
| 97 |
|
| 98 |
print("Vector store built successfully.")
|
| 99 |
|
|
|
|
| 100 |
|
|
|
|
|
|
|
| 101 |
if __name__ == "__main__":
|
| 102 |
build_vector_store()
|
|
|
|
| 1 |
import sqlite3
|
| 2 |
import chromadb
|
| 3 |
from sentence_transformers import SentenceTransformer
|
| 4 |
+
<<<<<<< HEAD
|
| 5 |
from pathlib import Path
|
| 6 |
import sys
|
| 7 |
sys.path.append(str(Path(__file__).parent.parent))
|
|
|
|
| 11 |
"""
|
| 12 |
Builds a persistent vector store from the data in the SQLite database,
|
| 13 |
embedding information about students, faculty, and courses.
|
| 14 |
+
=======
|
| 15 |
+
|
| 16 |
+
def build_vector_store():
|
| 17 |
+
"""
|
| 18 |
+
Builds a persistent vector store from the data in the SQLite database.
|
| 19 |
+
>>>>>>> feature/hf-deployment
|
| 20 |
"""
|
| 21 |
conn = sqlite3.connect('database/university.db')
|
| 22 |
cursor = conn.cursor()
|
| 23 |
|
| 24 |
+
<<<<<<< HEAD
|
| 25 |
documents = []
|
| 26 |
print("Creating student docs")
|
| 27 |
# === Build Student Documents ===
|
|
|
|
| 97 |
client = chromadb.PersistentClient(path="rag/vector_store")
|
| 98 |
collection = client.get_or_create_collection("university_data")
|
| 99 |
|
| 100 |
+
=======
|
| 101 |
+
# Fetch data from the database
|
| 102 |
+
students = cursor.execute("SELECT name, email FROM students").fetchall()
|
| 103 |
+
faculty = cursor.execute("SELECT name, email, department FROM faculty").fetchall()
|
| 104 |
+
courses = cursor.execute("SELECT name FROM courses").fetchall()
|
| 105 |
+
|
| 106 |
+
conn.close()
|
| 107 |
+
|
| 108 |
+
# Format data into documents
|
| 109 |
+
documents = []
|
| 110 |
+
for student in students:
|
| 111 |
+
documents.append(f"Student: {student[0]}, Email: {student[1]}")
|
| 112 |
+
for prof in faculty:
|
| 113 |
+
documents.append(f"Faculty: {prof[0]}, Email: {prof[1]}, Department: {prof[2]}")
|
| 114 |
+
for course in courses:
|
| 115 |
+
documents.append(f"Course: {course[0]}")
|
| 116 |
+
|
| 117 |
+
# Initialize the embedding model
|
| 118 |
+
model = SentenceTransformer('sentence-transformers/all-MiniLM-L6-v2')
|
| 119 |
+
|
| 120 |
+
# Create embeddings
|
| 121 |
+
embeddings = model.encode(documents)
|
| 122 |
+
|
| 123 |
+
# Initialize ChromaDB client and create a collection
|
| 124 |
+
client = chromadb.PersistentClient(path="/tmp/vector_store")
|
| 125 |
+
collection = client.get_or_create_collection("university_data")
|
| 126 |
+
|
| 127 |
+
# Add documents and embeddings to the collection
|
| 128 |
+
>>>>>>> feature/hf-deployment
|
| 129 |
collection.add(
|
| 130 |
embeddings=embeddings,
|
| 131 |
documents=documents,
|
|
|
|
| 134 |
|
| 135 |
print("Vector store built successfully.")
|
| 136 |
|
| 137 |
+
<<<<<<< HEAD
|
| 138 |
|
| 139 |
+
=======
|
| 140 |
+
>>>>>>> feature/hf-deployment
|
| 141 |
if __name__ == "__main__":
|
| 142 |
build_vector_store()
|
rag/retriever.py
CHANGED
|
@@ -1,9 +1,14 @@
|
|
| 1 |
import chromadb
|
| 2 |
from sentence_transformers import SentenceTransformer
|
| 3 |
import sqlite3
|
|
|
|
| 4 |
from helper import get_similarity_model, sanitize
|
| 5 |
|
| 6 |
def search(query: str, top_k: int = 10):
|
|
|
|
|
|
|
|
|
|
|
|
|
| 7 |
"""
|
| 8 |
Searches the vector store for the most relevant documents to a given query.
|
| 9 |
"""
|
|
@@ -15,6 +20,7 @@ def search(query: str, top_k: int = 10):
|
|
| 15 |
conn.close()
|
| 16 |
return [{"name": student[0]} for student in students]
|
| 17 |
|
|
|
|
| 18 |
# get SentenceTransformer model
|
| 19 |
model = get_similarity_model()
|
| 20 |
|
|
@@ -23,6 +29,16 @@ def search(query: str, top_k: int = 10):
|
|
| 23 |
|
| 24 |
# Initialize ChromaDB client and get the collection
|
| 25 |
client = chromadb.PersistentClient(path="rag/vector_store")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 26 |
collection = client.get_collection("university_data")
|
| 27 |
|
| 28 |
# Perform the search
|
|
@@ -30,9 +46,14 @@ def search(query: str, top_k: int = 10):
|
|
| 30 |
query_embeddings=query_embedding,
|
| 31 |
n_results=top_k
|
| 32 |
)
|
|
|
|
| 33 |
sanitized_context = [sanitize(doc) for doc in results['documents'][0]]
|
| 34 |
return sanitized_context
|
| 35 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 36 |
|
| 37 |
if __name__ == '__main__':
|
| 38 |
# Example usage
|
|
|
|
| 1 |
import chromadb
|
| 2 |
from sentence_transformers import SentenceTransformer
|
| 3 |
import sqlite3
|
| 4 |
+
<<<<<<< HEAD
|
| 5 |
from helper import get_similarity_model, sanitize
|
| 6 |
|
| 7 |
def search(query: str, top_k: int = 10):
|
| 8 |
+
=======
|
| 9 |
+
|
| 10 |
+
def search(query: str, top_k: int = 5):
|
| 11 |
+
>>>>>>> feature/hf-deployment
|
| 12 |
"""
|
| 13 |
Searches the vector store for the most relevant documents to a given query.
|
| 14 |
"""
|
|
|
|
| 20 |
conn.close()
|
| 21 |
return [{"name": student[0]} for student in students]
|
| 22 |
|
| 23 |
+
<<<<<<< HEAD
|
| 24 |
# get SentenceTransformer model
|
| 25 |
model = get_similarity_model()
|
| 26 |
|
|
|
|
| 29 |
|
| 30 |
# Initialize ChromaDB client and get the collection
|
| 31 |
client = chromadb.PersistentClient(path="rag/vector_store")
|
| 32 |
+
=======
|
| 33 |
+
# Initialize the embedding model
|
| 34 |
+
model = SentenceTransformer('sentence-transformers/all-MiniLM-L6-v2')
|
| 35 |
+
|
| 36 |
+
# Create the query embedding
|
| 37 |
+
query_embedding = model.encode([query])
|
| 38 |
+
|
| 39 |
+
# Initialize ChromaDB client and get the collection
|
| 40 |
+
client = chromadb.PersistentClient(path="/tmp/vector_store")
|
| 41 |
+
>>>>>>> feature/hf-deployment
|
| 42 |
collection = client.get_collection("university_data")
|
| 43 |
|
| 44 |
# Perform the search
|
|
|
|
| 46 |
query_embeddings=query_embedding,
|
| 47 |
n_results=top_k
|
| 48 |
)
|
| 49 |
+
<<<<<<< HEAD
|
| 50 |
sanitized_context = [sanitize(doc) for doc in results['documents'][0]]
|
| 51 |
return sanitized_context
|
| 52 |
|
| 53 |
+
=======
|
| 54 |
+
|
| 55 |
+
return results['documents'][0]
|
| 56 |
+
>>>>>>> feature/hf-deployment
|
| 57 |
|
| 58 |
if __name__ == '__main__':
|
| 59 |
# Example usage
|
requirements.txt
CHANGED
|
@@ -1,12 +1,11 @@
|
|
| 1 |
streamlit==1.37.0
|
| 2 |
python-dotenv==1.0.1
|
| 3 |
transformers
|
| 4 |
-
sentence-transformers
|
| 5 |
scikit-learn
|
| 6 |
-
|
| 7 |
sentence-transformers==5.1.0
|
| 8 |
numpy==1.26.4
|
| 9 |
huggingface-hub==0.34.4
|
| 10 |
-
chromadb
|
| 11 |
langdetect
|
| 12 |
-
nltk
|
|
|
|
| 1 |
streamlit==1.37.0
|
| 2 |
python-dotenv==1.0.1
|
| 3 |
transformers
|
|
|
|
| 4 |
scikit-learn
|
| 5 |
+
Faker==15.3.4
|
| 6 |
sentence-transformers==5.1.0
|
| 7 |
numpy==1.26.4
|
| 8 |
huggingface-hub==0.34.4
|
| 9 |
+
chromadb==1.0.21
|
| 10 |
langdetect
|
| 11 |
+
nltk
|
startup.sh
ADDED
|
@@ -0,0 +1,4 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
|
| 3 |
+
# Run the Streamlit app
|
| 4 |
+
streamlit run app.py
|