Samuel Oberhofer commited on
Commit
760d27d
·
2 Parent(s): 58f91d2eddbff5

Merge feature/hf-deployment into master

Browse files
.github/workflows/push-to-hf-space.yml ADDED
@@ -0,0 +1,37 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ name: Push to Hugging Face Space
2
+
3
+ on:
4
+ push:
5
+ branches: [ feature/hf-deployment ] # change if you use another default branch
6
+
7
+ jobs:
8
+ sync:
9
+ runs-on: ubuntu-latest
10
+
11
+ steps:
12
+ - name: Check out repo
13
+ uses: actions/checkout@v4
14
+ with:
15
+ fetch-depth: 0 # keep history; helpful for subtree/sparse etc.
16
+
17
+ - name: Set up Git LFS (if you track large files)
18
+ run: |
19
+ sudo apt-get update
20
+ sudo apt-get install -y git-lfs
21
+ git lfs install
22
+
23
+ - name: Push to Space
24
+ env:
25
+ HF_TOKEN: ${{ secrets.HF_TOKEN }}
26
+ HF_SPACE: pxdelta/RAG # or use ${{ secrets.HF_SPACE }}
27
+ run: |
28
+ # Configure git identity for the CI push
29
+ git config user.name "github-actions[bot]"
30
+ git config user.email "github-actions[bot]@users.noreply.github.com"
31
+
32
+ # Add the Space as an authenticated remote.
33
+ # Username can be anything (often 'oauth2' or your HF username); token goes in the password slot.
34
+ git remote add hf "https://oauth2:${HF_TOKEN}@huggingface.co/spaces/${HF_SPACE}"
35
+
36
+ # Push the current GitHub branch (e.g., main) to the Space's main branch
37
+ git push hf HEAD:main --force
.gitignore CHANGED
@@ -2,10 +2,6 @@
2
  .env
3
  venv/
4
  env/
5
- .venv
6
-
7
- notes.txt
8
- test.ipynb
9
 
10
  # Python
11
  __pycache__/
@@ -22,6 +18,3 @@ Thumbs.db
22
  # Generated files
23
  database/university.db
24
  rag/vector_store/
25
-
26
-
27
- secrets_local.py
 
2
  .env
3
  venv/
4
  env/
 
 
 
 
5
 
6
  # Python
7
  __pycache__/
 
18
  # Generated files
19
  database/university.db
20
  rag/vector_store/
 
 
 
Dockerfile ADDED
@@ -0,0 +1,29 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Use the official Python image
2
+ FROM python:3.10-slim
3
+
4
+ # Create a non-root user
5
+ RUN useradd -m -u 1000 user
6
+
7
+ # Set the working directory
8
+ WORKDIR /app
9
+
10
+ # Copy all application files
11
+ COPY . .
12
+
13
+ # Grant ownership of the app directory to the user
14
+ RUN chown -R user:user /app
15
+
16
+ # Switch to the non-root user
17
+ USER user
18
+
19
+ # Install dependencies
20
+ RUN pip install --user -r requirements.txt
21
+
22
+ # Make the startup script executable
23
+ RUN chmod +x startup.sh
24
+
25
+ # Add the user's local bin directory to the PATH
26
+ ENV PATH="/home/user/.local/bin:${PATH}"
27
+
28
+ # Set the entrypoint to the startup script
29
+ CMD ["./startup.sh"]
guards/input.py ADDED
@@ -0,0 +1,30 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import re
2
+
3
+ def is_valid(query: str) -> bool:
4
+ """
5
+ Validates the user's query.
6
+ """
7
+ # Check for query length
8
+ if len(query) > 500:
9
+ return False
10
+
11
+ # Check for SQL injection patterns
12
+ sql_injection_patterns = [
13
+ r"(\s*(--|#|;))",
14
+ r"(\s*(union|select|insert|update|delete|drop|alter)\s+)",
15
+ ]
16
+ for pattern in sql_injection_patterns:
17
+ if re.search(pattern, query, re.IGNORECASE):
18
+ return False
19
+
20
+ return True
21
+
22
+ if __name__ == '__main__':
23
+ # Example usage
24
+ valid_query = "What are the computer science courses?"
25
+ invalid_query_long = "a" * 501
26
+ invalid_query_sql = "SELECT * FROM students;"
27
+
28
+ print(f"'{valid_query}' is valid: {is_valid(valid_query)}")
29
+ print(f"'long query' is valid: {is_valid(invalid_query_long)}")
30
+ print(f"'{invalid_query_sql}' is valid: {is_valid(invalid_query_sql)}")
guards/output.py ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import re
2
+
3
+ def sanitize(response: str) -> str:
4
+ """
5
+ Sanitizes the LLM's response.
6
+ """
7
+ # Redact email addresses
8
+ response = re.sub(r"[a-zA-Z0-9_.+-]+@[a-zA-Z0-9-]+\.[a-zA-Z0-9-.]+", "[REDACTED EMAIL]", response)
9
+
10
+ # Check for hallucination patterns
11
+ hallucination_patterns = [
12
+ r"as an ai language model",
13
+ r"i cannot answer that question",
14
+ ]
15
+ for pattern in hallucination_patterns:
16
+ if re.search(pattern, response, re.IGNORECASE):
17
+ return "I'm sorry, I don't have enough information to answer that question."
18
+
19
+ return response
20
+
21
+ if __name__ == '__main__':
22
+ # Example usage
23
+ valid_response = "Professor Smith's email is john.smith@university.edu."
24
+ hallucination_response = "As an AI language model, I cannot provide that information."
25
+
26
+ print(f"Original: '{valid_response}'\nSanitized: '{sanitize(valid_response)}'")
27
+ print(f"\nOriginal: '{hallucination_response}'\nSanitized: '{sanitize(hallucination_response)}'")
rag/build_vector_store.py CHANGED
@@ -1,6 +1,7 @@
1
  import sqlite3
2
  import chromadb
3
  from sentence_transformers import SentenceTransformer
 
4
  from pathlib import Path
5
  import sys
6
  sys.path.append(str(Path(__file__).parent.parent))
@@ -10,10 +11,17 @@ def build_vector_store():
10
  """
11
  Builds a persistent vector store from the data in the SQLite database,
12
  embedding information about students, faculty, and courses.
 
 
 
 
 
 
13
  """
14
  conn = sqlite3.connect('database/university.db')
15
  cursor = conn.cursor()
16
 
 
17
  documents = []
18
  print("Creating student docs")
19
  # === Build Student Documents ===
@@ -89,6 +97,35 @@ def build_vector_store():
89
  client = chromadb.PersistentClient(path="rag/vector_store")
90
  collection = client.get_or_create_collection("university_data")
91
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
92
  collection.add(
93
  embeddings=embeddings,
94
  documents=documents,
@@ -97,6 +134,9 @@ def build_vector_store():
97
 
98
  print("Vector store built successfully.")
99
 
 
100
 
 
 
101
  if __name__ == "__main__":
102
  build_vector_store()
 
1
  import sqlite3
2
  import chromadb
3
  from sentence_transformers import SentenceTransformer
4
+ <<<<<<< HEAD
5
  from pathlib import Path
6
  import sys
7
  sys.path.append(str(Path(__file__).parent.parent))
 
11
  """
12
  Builds a persistent vector store from the data in the SQLite database,
13
  embedding information about students, faculty, and courses.
14
+ =======
15
+
16
+ def build_vector_store():
17
+ """
18
+ Builds a persistent vector store from the data in the SQLite database.
19
+ >>>>>>> feature/hf-deployment
20
  """
21
  conn = sqlite3.connect('database/university.db')
22
  cursor = conn.cursor()
23
 
24
+ <<<<<<< HEAD
25
  documents = []
26
  print("Creating student docs")
27
  # === Build Student Documents ===
 
97
  client = chromadb.PersistentClient(path="rag/vector_store")
98
  collection = client.get_or_create_collection("university_data")
99
 
100
+ =======
101
+ # Fetch data from the database
102
+ students = cursor.execute("SELECT name, email FROM students").fetchall()
103
+ faculty = cursor.execute("SELECT name, email, department FROM faculty").fetchall()
104
+ courses = cursor.execute("SELECT name FROM courses").fetchall()
105
+
106
+ conn.close()
107
+
108
+ # Format data into documents
109
+ documents = []
110
+ for student in students:
111
+ documents.append(f"Student: {student[0]}, Email: {student[1]}")
112
+ for prof in faculty:
113
+ documents.append(f"Faculty: {prof[0]}, Email: {prof[1]}, Department: {prof[2]}")
114
+ for course in courses:
115
+ documents.append(f"Course: {course[0]}")
116
+
117
+ # Initialize the embedding model
118
+ model = SentenceTransformer('sentence-transformers/all-MiniLM-L6-v2')
119
+
120
+ # Create embeddings
121
+ embeddings = model.encode(documents)
122
+
123
+ # Initialize ChromaDB client and create a collection
124
+ client = chromadb.PersistentClient(path="/tmp/vector_store")
125
+ collection = client.get_or_create_collection("university_data")
126
+
127
+ # Add documents and embeddings to the collection
128
+ >>>>>>> feature/hf-deployment
129
  collection.add(
130
  embeddings=embeddings,
131
  documents=documents,
 
134
 
135
  print("Vector store built successfully.")
136
 
137
+ <<<<<<< HEAD
138
 
139
+ =======
140
+ >>>>>>> feature/hf-deployment
141
  if __name__ == "__main__":
142
  build_vector_store()
rag/retriever.py CHANGED
@@ -1,9 +1,14 @@
1
  import chromadb
2
  from sentence_transformers import SentenceTransformer
3
  import sqlite3
 
4
  from helper import get_similarity_model, sanitize
5
 
6
  def search(query: str, top_k: int = 10):
 
 
 
 
7
  """
8
  Searches the vector store for the most relevant documents to a given query.
9
  """
@@ -15,6 +20,7 @@ def search(query: str, top_k: int = 10):
15
  conn.close()
16
  return [{"name": student[0]} for student in students]
17
 
 
18
  # get SentenceTransformer model
19
  model = get_similarity_model()
20
 
@@ -23,6 +29,16 @@ def search(query: str, top_k: int = 10):
23
 
24
  # Initialize ChromaDB client and get the collection
25
  client = chromadb.PersistentClient(path="rag/vector_store")
 
 
 
 
 
 
 
 
 
 
26
  collection = client.get_collection("university_data")
27
 
28
  # Perform the search
@@ -30,9 +46,14 @@ def search(query: str, top_k: int = 10):
30
  query_embeddings=query_embedding,
31
  n_results=top_k
32
  )
 
33
  sanitized_context = [sanitize(doc) for doc in results['documents'][0]]
34
  return sanitized_context
35
 
 
 
 
 
36
 
37
  if __name__ == '__main__':
38
  # Example usage
 
1
  import chromadb
2
  from sentence_transformers import SentenceTransformer
3
  import sqlite3
4
+ <<<<<<< HEAD
5
  from helper import get_similarity_model, sanitize
6
 
7
  def search(query: str, top_k: int = 10):
8
+ =======
9
+
10
+ def search(query: str, top_k: int = 5):
11
+ >>>>>>> feature/hf-deployment
12
  """
13
  Searches the vector store for the most relevant documents to a given query.
14
  """
 
20
  conn.close()
21
  return [{"name": student[0]} for student in students]
22
 
23
+ <<<<<<< HEAD
24
  # get SentenceTransformer model
25
  model = get_similarity_model()
26
 
 
29
 
30
  # Initialize ChromaDB client and get the collection
31
  client = chromadb.PersistentClient(path="rag/vector_store")
32
+ =======
33
+ # Initialize the embedding model
34
+ model = SentenceTransformer('sentence-transformers/all-MiniLM-L6-v2')
35
+
36
+ # Create the query embedding
37
+ query_embedding = model.encode([query])
38
+
39
+ # Initialize ChromaDB client and get the collection
40
+ client = chromadb.PersistentClient(path="/tmp/vector_store")
41
+ >>>>>>> feature/hf-deployment
42
  collection = client.get_collection("university_data")
43
 
44
  # Perform the search
 
46
  query_embeddings=query_embedding,
47
  n_results=top_k
48
  )
49
+ <<<<<<< HEAD
50
  sanitized_context = [sanitize(doc) for doc in results['documents'][0]]
51
  return sanitized_context
52
 
53
+ =======
54
+
55
+ return results['documents'][0]
56
+ >>>>>>> feature/hf-deployment
57
 
58
  if __name__ == '__main__':
59
  # Example usage
requirements.txt CHANGED
@@ -1,12 +1,11 @@
1
  streamlit==1.37.0
2
  python-dotenv==1.0.1
3
  transformers
4
- sentence-transformers
5
  scikit-learn
6
- faker
7
  sentence-transformers==5.1.0
8
  numpy==1.26.4
9
  huggingface-hub==0.34.4
10
- chromadb
11
  langdetect
12
- nltk
 
1
  streamlit==1.37.0
2
  python-dotenv==1.0.1
3
  transformers
 
4
  scikit-learn
5
+ Faker==15.3.4
6
  sentence-transformers==5.1.0
7
  numpy==1.26.4
8
  huggingface-hub==0.34.4
9
+ chromadb==1.0.21
10
  langdetect
11
+ nltk
startup.sh ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ #!/bin/bash
2
+
3
+ # Run the Streamlit app
4
+ streamlit run app.py