kotlarmilos commited on
Commit
3562e2e
Β·
verified Β·
1 Parent(s): 12d9d49

Upload 2 files

Browse files
Files changed (2) hide show
  1. app.py +136 -29
  2. requirements.txt +51 -1
app.py CHANGED
@@ -1,12 +1,135 @@
 
 
 
 
 
 
 
 
1
  import gradio as gr
2
- from huggingface_hub import InferenceClient
3
 
4
- """
5
- For more information on `huggingface_hub` Inference API support, please check the docs: https://huggingface.co/docs/huggingface_hub/v0.22.2/en/guides/inference
6
- """
7
- client = InferenceClient("HuggingFaceH4/zephyr-7b-beta")
8
 
9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
10
  def respond(
11
  message,
12
  history: list[tuple[str, str]],
@@ -15,30 +138,15 @@ def respond(
15
  temperature,
16
  top_p,
17
  ):
18
- messages = [{"role": "system", "content": system_message}]
19
-
20
- for val in history:
21
- if val[0]:
22
- messages.append({"role": "user", "content": val[0]})
23
- if val[1]:
24
- messages.append({"role": "assistant", "content": val[1]})
25
-
26
- messages.append({"role": "user", "content": message})
27
-
28
- response = ""
29
-
30
- for message in client.chat_completion(
31
- messages,
32
- max_tokens=max_tokens,
33
- stream=True,
34
- temperature=temperature,
35
- top_p=top_p,
36
- ):
37
- token = message.choices[0].delta.content
38
-
39
- response += token
40
- yield response
41
 
 
 
 
42
 
43
  """
44
  For information on how to customize the ChatInterface, peruse the gradio docs: https://www.gradio.app/docs/chatinterface
@@ -59,6 +167,5 @@ demo = gr.ChatInterface(
59
  ],
60
  )
61
 
62
-
63
  if __name__ == "__main__":
64
  demo.launch()
 
1
+ import os
2
+ import json
3
+ import faiss
4
+ from pathlib import Path
5
+ from git import Repo
6
+ from huggingface_hub import snapshot_download
7
+ from sentence_transformers import SentenceTransformer
8
+ from openai import OpenAI
9
  import gradio as gr
10
+ from openai import AzureOpenAI
11
 
 
 
 
 
12
 
13
 
14
+ # β€”β€”β€” Configuration β€”β€”β€”
15
+ REPO_URL = "https://github.com/dotnet/xharness.git"
16
+ REPO_LOCAL_DIR = Path("artifacts/repo_code")
17
+ HF_REPO_ID = "kotlarmilos/repository-learning"
18
+ HF_BASE_DIR = Path("artifacts/repo_hf")
19
+ HF_INDEX_DIR = HF_BASE_DIR / "dotnet-xharness" / "index"
20
+ METADATA_PATH = HF_INDEX_DIR / "metadata.json"
21
+ ROOT_DIR = REPO_LOCAL_DIR # where your repo code lives
22
+ AZURE_OPENAI_ENDPOINT = os.getenv("AZURE_OPENAI_ENDPOINT")
23
+ AZURE_OPENAI_API_KEY = os.getenv("AZURE_OPENAI_API_KEY")
24
+ API_VERSION = "2024-12-01-preview"
25
+ EMBEDDER_MODEL = "sentence-transformers/all-MiniLM-L6-v2"
26
+ OPENAI_MODEL = "gpt-4o-mini"
27
+ TOP_K = 5
28
+
29
+
30
+
31
+ # β€”β€”β€” Step 1: Acquire code and artifacts β€”β€”β€”
32
+ # Clone or pull the GitHub repo
33
+ if REPO_LOCAL_DIR.exists() and (REPO_LOCAL_DIR / ".git").exists():
34
+ Repo(REPO_LOCAL_DIR).remotes.origin.pull()
35
+ else:
36
+ Repo.clone_from(REPO_URL, REPO_LOCAL_DIR)
37
+
38
+ # Download Hugging Face snapshots for index & metadata
39
+ snapshot_download(
40
+ repo_id=HF_REPO_ID,
41
+ local_dir=str(HF_BASE_DIR),
42
+ local_dir_use_symlinks=False,
43
+ token=os.getenv("HUGGINGFACE_HUB_TOKEN"),
44
+ )
45
+
46
+ # β€”β€”β€” Step 2: Load FAISS index & metadata β€”β€”β€”
47
+ index = faiss.read_index(str(HF_INDEX_DIR / "index.faiss"))
48
+ with open(METADATA_PATH, "r", encoding="utf-8") as f:
49
+ metadata = json.load(f)
50
+
51
+ # β€”β€”β€” Step 3: Prepare embedder & OpenAI client β€”β€”β€”
52
+ embedder = SentenceTransformer(EMBEDDER_MODEL)
53
+ openai = AzureOpenAI(
54
+ api_version=API_VERSION,
55
+ azure_endpoint=AZURE_OPENAI_ENDPOINT,
56
+ api_key=AZURE_OPENAI_API_KEY,
57
+ )
58
+
59
+ # β€”β€”β€” Helper: load code snippets by FAISS ID β€”β€”β€”
60
+ def load_snippets_from_metadata(ids, metadata, root_dir):
61
+ snippets = []
62
+ for idx in ids:
63
+ entry = metadata[idx]
64
+ file_rel = entry["file"]
65
+ start, end = entry["start_line"], entry["end_line"]
66
+ file_path = Path(root_dir) / file_rel
67
+ try:
68
+ lines = file_path.read_text(encoding="utf-8").splitlines()
69
+ code = "\n".join(lines[start-1 : end]).rstrip()
70
+ except FileNotFoundError:
71
+ code = f"# ERROR: {file_rel} not found"
72
+ snippets.append({
73
+ "file": file_rel,
74
+ "lines": (start, end),
75
+ "code": code,
76
+ "description": entry.get("llm_description", "")
77
+ })
78
+ return snippets
79
+
80
+ # β€”β€”β€” Core: embed question, retrieve, and call OpenAI β€”β€”β€”
81
+ def answer_from_index(question: str, top_k: int = TOP_K) -> str:
82
+ # 1) Encode question
83
+ q_emb = embedder.encode([question])
84
+ # 2) Search FAISS
85
+ _, indices = index.search(q_emb, top_k)
86
+ ids = indices[0].tolist()
87
+ # 3) Load code snippets
88
+ snippets = load_snippets_from_metadata(ids, metadata, ROOT_DIR)
89
+ # 4) Build context block
90
+ context_parts = []
91
+ for snip in snippets:
92
+ context_parts.append(
93
+ f"File: {snip['file']} (lines {snip['lines'][0]}–{snip['lines'][1]})\n"
94
+ "```python\n"
95
+ f"{snip['code']}\n"
96
+ "```\n"
97
+ f"Description: {snip['description']}"
98
+ )
99
+ context_block = "\n\n".join(context_parts)
100
+ # 5) Prompt OpenAI
101
+ prompt = (
102
+ "You are a code assistant. Use the following code snippets to answer the user's question.\n\n"
103
+ f"{context_block}\n\n"
104
+ "Question:\n" f"{question}\n\n"
105
+ "Answer:"
106
+ )
107
+ resp = openai.chat.completions.create(
108
+ model=OPENAI_MODEL,
109
+ messages=[{"role": "user", "content": prompt}],
110
+ temperature=0.2,
111
+ max_tokens=512
112
+ )
113
+ return resp.choices[0].message.content.strip()
114
+
115
+ def rewrite_followup(history: list[tuple[str,str]], followup: str) -> str:
116
+ # history is list of (user,assistant) pairs
117
+ convo = "\n".join(
118
+ f"User: {u}\nAssistant: {a}" for u,a in history[-4:]
119
+ )
120
+ prompt = (
121
+ "Given the conversation below, rewrite the final user query into "
122
+ "a standalone question.\n\n"
123
+ f"{convo}\nUser: {followup}\n\nStandalone question:"
124
+ )
125
+ resp = openai.chat.completions.create(
126
+ model=OPENAI_MODEL,
127
+ messages=[{"role":"user","content":prompt}],
128
+ temperature=0,
129
+ max_tokens=128
130
+ )
131
+ return resp.choices[0].message.content.strip()
132
+
133
  def respond(
134
  message,
135
  history: list[tuple[str, str]],
 
138
  temperature,
139
  top_p,
140
  ):
141
+ # Use the existing answer_from_index function to get the response
142
+
143
+ standalone = rewrite_followup(history, message)
144
+ response = answer_from_index(standalone)
145
+ yield response
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
146
 
147
+ # β€”β€”β€” Gradio interface β€”β€”β€”
148
+ def on_submit(question: str) -> str:
149
+ return answer_from_index(question)
150
 
151
  """
152
  For information on how to customize the ChatInterface, peruse the gradio docs: https://www.gradio.app/docs/chatinterface
 
167
  ],
168
  )
169
 
 
170
  if __name__ == "__main__":
171
  demo.launch()
requirements.txt CHANGED
@@ -1 +1,51 @@
1
- huggingface_hub==0.25.2
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Core ML dependencies
2
+ torch>=2.0.0
3
+ transformers>=4.30.0
4
+ sentence-transformers>=2.2.2
5
+ faiss-cpu>=1.7.4
6
+ numpy>=1.21.0
7
+
8
+ # Data processing
9
+ pandas>=1.5.0
10
+ datasets>=2.12.0
11
+ accelerate>=0.20.0
12
+ peft>=0.4.0
13
+ bitsandbytes>=0.39.0
14
+ huggingface_hub>=0.16.0
15
+
16
+ # GitHub integration
17
+ PyGithub
18
+ GitPython>=3.1.0
19
+ requests>=2.28.0
20
+ python-dotenv>=1.0.0
21
+ tenacity>=8.2.0
22
+
23
+ # Tree-sitter parsers
24
+ tree-sitter>=0.20.0
25
+ tree-sitter-python>=0.20.0
26
+ tree-sitter-c>=0.20.0
27
+ tree-sitter-cpp>=0.20.0
28
+ tree-sitter-java>=0.20.0
29
+ tree-sitter-c-sharp>=0.20.0
30
+ tree-sitter-javascript>=0.20.0
31
+
32
+ # Web interface
33
+ flask>=2.3.0
34
+ flask-session>=0.5.0
35
+
36
+ # Utilities
37
+ tqdm>=4.64.0
38
+ schedule>=1.2.0
39
+ openai
40
+
41
+ # Development
42
+ pytest>=7.0.0
43
+ black>=23.0.0
44
+ flake8>=6.0.0
45
+
46
+ # Transformers library for NLP tasks
47
+ transformers
48
+ matplotlib
49
+ scikit-learn
50
+ gradio
51
+ openai