Sk4467 commited on
Commit
f34a2c8
·
1 Parent(s): 480a195

Initial Commit

Browse files
Files changed (3) hide show
  1. app.py +302 -0
  2. custom_prompt_template.py +23 -0
  3. requirements.txt +4 -0
app.py ADDED
@@ -0,0 +1,302 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import streamlit as st
2
+ import requests
3
+ import justext
4
+ import pdfplumber
5
+ import docx2txt
6
+ import test_prompt as prompt
7
+ import json
8
+ import os
9
+ import openai
10
+
11
+ from custom_prompt_template import InstructionGenerationTemplate
12
+
13
+
14
+ st.set_page_config(page_title="LLM instruction")
15
+
16
+ st.sidebar.success("Select a page above")
17
+
18
+
19
+ # function for the odia stoplists justext
20
+ def odia_stoplist():
21
+ odia_stopwords = [
22
+ "ଏହି", "ଏକ", "ଏକାଉଣଟ", "ମୁଁ", "ମୋର", "ମୁଁ ନିଜେ", "ଆମେ", "ଆମର", "ଆମର", "ଆମେ ନିଜେ", "ତୁମେ", "ତୁମର", "ତୁମର",
23
+ "ନିଜେ", "ନିଜେ", "ସେ", "ତାଙ୍କୁ", "ତାଙ୍କର",
24
+ "ନିଜେ", "ସେ", "ତାଙ୍କୁ", "ତାଙ୍କର", "ନିଜେ", "ଏହା", "ଏହାର", "ନିଜେ |", "ସେମାନେ", "ସେଗୁଡିକ", "ସେମାନଙ୍କର",
25
+ "ସେମାନଙ୍କର", "ନିଜେ |", "କଣ", "ଯାହା", "କିଏ", "କାହାକୁ",
26
+ "ଏହା", "ତାହା", "ଏଗୁଡ଼ିକ", "ସେଗୁଡ଼ିକ", "ମୁଁ", "ହେଉଛି", "ହେଉଛି |", "ଥିଲା", "ଥିଲା |", "ହୁଅ", "ହୋଇସାରିଛି |", "ହେବା",
27
+ "ଅଛି", "ଅଛି", "ଥିଲା", "ଅଛି", "କର", "କରେ |",
28
+ "କରିଛନ୍ତି", "କରିବା", "ଏବଂ", "କିନ୍ତୁ", "ଯଦି", "କିମ୍ବା", "କାରଣ", "ଯେପରି", "ପର୍ଯ୍ୟନ୍ତ", "ଯେତେବେଳେ", "ର", "ପାଇଁ",
29
+ "ସହିତ", "ବିଷୟରେ", "ବିପକ୍ଷରେ", "ମଧ୍ୟରେ", "ଭିତରକୁ", "ମାଧ୍ୟମରେ",
30
+ "ସମୟରେ", "ପୂର୍ବରୁ", "ପରେ", "ଉପରେ", "ନିମ୍ନରେ |", "କୁ", "ଠାରୁ", "ଅପ୍", "ତଳକୁ", "ଭିତରେ", "ବାହାରେ", "ଉପରେ", "ବନ୍ଦ",
31
+ "ସମାପ୍ତ", "ତଳେ |", "ପୁନର୍ବାର", "ଆଗକୁ",
32
+ "ତାପରେ", "ଥରେ |", "ଏଠାରେ", "ସେଠାରେ", "କେବେ", "କେଉଁଠାରେ", "କିପରି", "ସମସ୍ତ", "ଉଭୟ", "ପ୍ରତ୍ୟେକ", "ଅଳ୍ପ", "ଅଧିକ",
33
+ "ଅଧିକାଂଶ", "ଅନ୍ୟ", "କେତେକ", "ଏହିପରି",
34
+ "ନୁହେଁ |", "କେବଳ", "ନିଜର", "ସମାନ", "ତେଣୁ", "ଅପେକ୍ଷା", "ମଧ୍ୟ", "ବହୁତ", "କରିପାରିବେ |", "ଇଚ୍ଛା", "କେବଳ",
35
+ "କରିବା ଉଚିତ", "ବର୍ତ୍ତମାନ"
36
+ ]
37
+ return frozenset(odia_stopwords)
38
+
39
+
40
+ # function to extract data from url using justext
41
+ def extract_data_from_url(url, language):
42
+ try:
43
+ response = requests.get(url)
44
+ response.raise_for_status()
45
+ page = response.content
46
+
47
+ para = ""
48
+ if language == "English":
49
+ paragraphs = justext.justext(page, justext.get_stoplist("English"))
50
+ elif language == "Hindi":
51
+ paragraphs = justext.justext(page, justext.get_stoplist("Hindi"))
52
+ elif language == "Odia":
53
+ paragraphs = justext.justext(
54
+ page, odia_stoplist(), 70, 140, 0.0, 0.02, 0.5, 150, False
55
+ )
56
+
57
+ for paragraph in paragraphs:
58
+ if not paragraph.is_boilerplate:
59
+ para = para + "\n" + paragraph.text
60
+ # returning the extracted data i.e para as string
61
+ return para
62
+ except Exception as e:
63
+ st.error(e)
64
+
65
+
66
+ # function to extract data from documents
67
+ def extract_data_from_documents(documents):
68
+ data = ""
69
+ if documents is not None:
70
+ for document in documents:
71
+ document_details = {
72
+ "filename": document.name,
73
+ "filetype": document.type,
74
+ "filesize": document.size,
75
+ }
76
+ st.write(document_details)
77
+
78
+ # Extract content from the txt file
79
+ if document.type == "text/plain":
80
+ # Read as bytes
81
+ data += str(document.read(), "utf-8")
82
+
83
+ # Extract content from the pdf file
84
+ elif document.type == "application/pdf":
85
+ # using pdfplumber
86
+ try:
87
+ with pdfplumber.open(document) as pdf:
88
+ all_text = ""
89
+ for page in pdf.pages:
90
+ text = page.extract_text()
91
+ all_text += text + "\n"
92
+ data += all_text
93
+ except requests.exceptions.RequestException as e:
94
+ st.write("None")
95
+
96
+ # Extract content from the docx file
97
+ elif (
98
+ document.type
99
+ == "application/vnd.openxmlformats-officedocument.wordprocessingml.document"
100
+ ):
101
+ data += docx2txt.process(document)
102
+
103
+ # return extract data
104
+ return data
105
+ else:
106
+ st.error("Error: An error occurred while fetching content.")
107
+ # return extract status, and the data extracted
108
+ return None
109
+
110
+
111
+ # function for the keyboard
112
+
113
+
114
+
115
+ # Check the inputs for language, promptType
116
+ def valid_drop_down(language, promptType, noOfQuestions, instructionFormat):
117
+ langFlag = False
118
+ promptFlag = False
119
+ noOfQuestionFlag = False
120
+ instructionFormatFlag = False
121
+
122
+ if language:
123
+ langFlag = True
124
+ if promptType:
125
+ promptFlag = True
126
+ if noOfQuestions:
127
+ noOfQuestionFlag = True
128
+ if instructionFormat:
129
+ instructionFormatFlag = True
130
+ # checking for the compalsory inputs and return true only if all are set
131
+ return langFlag & promptFlag & noOfQuestionFlag & instructionFormatFlag
132
+
133
+
134
+ def main():
135
+ # setting up the initial session_states
136
+ if "extract_button" not in st.session_state:
137
+ st.session_state.extract_button = False
138
+ if "submit" not in st.session_state:
139
+ st.session_state.submit = False
140
+ if "generated" not in st.session_state:
141
+ st.session_state.generated = False
142
+
143
+ st.subheader("LLM Instructions")
144
+
145
+ # form to get the inputs
146
+ with st.form(key="form1"):
147
+ st.write("#")
148
+
149
+ # dropdown for language
150
+ language = st.selectbox("Select a language", ("", "English", "Hindi", "Odia"))
151
+
152
+ # dropdown for prompt type
153
+ promptType = st.selectbox(
154
+ "Select the Prompt type", ("", "Input text", "Url", "Document")
155
+ )
156
+ # inputs for number
157
+ noOfQuestions = st.number_input(
158
+ "Number of questions to generate:", min_value=1, max_value=20, value=10
159
+ )
160
+
161
+ # dropdown for language
162
+ instructionFormat = st.selectbox(
163
+ "Format of instruction:", ("Imperative sentence", "Question")
164
+ )
165
+
166
+ # checkbox for additional info bool val
167
+ addInfoCheckbox = st.checkbox("Input Additional Instructions", value=False)
168
+
169
+ st.write("##")
170
+
171
+ # form submit button and setting up the session_state
172
+ if st.form_submit_button():
173
+ st.session_state.submit = True
174
+
175
+ if st.session_state.submit:
176
+ # extends the prompt form to extract the data
177
+ with st.expander(label="prompt"):
178
+ with st.form(key="form2"):
179
+ # calling the function inside if to check valid drop down inputs
180
+ if valid_drop_down(
181
+ language, promptType, noOfQuestions, instructionFormat
182
+ ):
183
+ if promptType == "Input text":
184
+ inputText = st.text_area(
185
+ label="For Instructions",
186
+ placeholder="Please enter your text here",
187
+ )
188
+
189
+ elif promptType == "Url":
190
+ url = st.text_input(
191
+ label="For URL", placeholder="Please enter your text here"
192
+ )
193
+ elif promptType == "Document":
194
+ documents = st.file_uploader(
195
+ label="For Documents ( pdf / txt / docx )",
196
+ type=["pdf", "txt", "docx"],
197
+ accept_multiple_files=True,
198
+ )
199
+
200
+ if addInfoCheckbox:
201
+ additionalInfo = st.text_input(
202
+ label="Additional Instructions",
203
+ placeholder="Please enter your text here",
204
+ )
205
+
206
+ if st.form_submit_button():
207
+ st.session_state.extract_button = True
208
+ # st.experimental_rerun()
209
+
210
+ # extracting data
211
+ if st.session_state.extract_button:
212
+ # extracting data
213
+ if promptType == "Input text":
214
+ extractedData = inputText
215
+
216
+ elif promptType == "Url":
217
+ extractedURLData = extract_data_from_url(url, language)
218
+ extractedData = extractedURLData
219
+
220
+ elif promptType == "Document":
221
+ if not documents:
222
+ documents = None
223
+ else:
224
+ for doc in documents:
225
+ if doc.name.split(".")[-1].lower() not in ["pdf", "txt", "docx"]:
226
+ # if documents is not the relevant type
227
+ st.error("Unsupported file: " + doc.name)
228
+
229
+ extractedDocumentData = extract_data_from_documents(documents)
230
+ extractedData = extractedDocumentData
231
+
232
+ # if the values are extracted running the custom prompt by creating an instance
233
+ if extractedData:
234
+ # ----------------------------- RUNNING THE PROMPT -----------------------------
235
+
236
+ # running the prompt form here
237
+ openai.api_key = os.getenv("OPENAI_API_KEY")
238
+ my_prompt_template = InstructionGenerationTemplate()
239
+
240
+ # providing the rules for the instructions to be generated
241
+ additional_rules = """
242
+ - You do not need to provide a response to the generated examples.
243
+ - You must return the response in the specified language.
244
+ - Each generated instruction can be either an imperative sentence or a question.
245
+ - Return the result in dictionary , where the key is the serial number and the value is an instruction.
246
+ """
247
+
248
+ if st.button("Generate Instructions"):
249
+ prompt = my_prompt_template.format(
250
+ num_questions=noOfQuestions,
251
+ context=extractedData,
252
+ instruction_format=instructionFormat,
253
+ lang=language,
254
+ additional_rules=additional_rules
255
+ )
256
+ response = openai.ChatCompletion.create(
257
+ model="gpt-3.5-turbo",
258
+ messages=[
259
+ {"role": "system", "content": prompt},
260
+ # {"role": "user", "content": f"Generate {num_questions} diverse questions based on the context provided below in the form of {instruction_format}.\n\n{context}\n\n{additional_rules}"},
261
+ ])
262
+
263
+ if "result" not in st.session_state:
264
+ st.session_state["result"] = response.choices[0].message.content
265
+ st.session_state.generated = True
266
+
267
+ if st.session_state.generated:
268
+ # displaying the generated instructions
269
+ st.write("Generated Insuctions")
270
+ # st.write(result)
271
+ result = st.session_state["result"]
272
+ print(type(result))
273
+ print(result)
274
+ # cleaned_result=result.replace("<example>","").replace("</example>","").replace("\\n","\n")
275
+ result_dict=json.loads(result)
276
+ print(type(result_dict))
277
+ print(result_dict)
278
+ # print(result_dict)
279
+ # for key,value in result_dict.items():
280
+ # print(f"{key}:{value}")
281
+ # including the questions as checkboxes for generating further solutions
282
+ # Creating list to display the selected instructions
283
+ selected_items = [f"{value} " for key, value in result_dict.items() if st.checkbox(f"Q{key} : {value}")]
284
+
285
+ # Display the selected items as a list
286
+ if selected_items:
287
+ st.write("Selected Items:")
288
+ st.write(selected_items)
289
+ else:
290
+ st.write("No items selected.")
291
+
292
+
293
+ if st.button("clear"):
294
+ st.session_state.extract_button = False
295
+ st.session_state.submit = False
296
+ st.session_state.generated = True
297
+ del st.session_state["result"]
298
+ st.experimental_rerun()
299
+
300
+
301
+ if __name__ == "__main__":
302
+ main()
custom_prompt_template.py ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from typing import List
2
+ import langchain
3
+ class InstructionGenerationTemplate(langchain.prompts.PromptTemplate):
4
+ """A custom prompt template for generating instructions."""
5
+
6
+ input_variables: List[str] = ["num_questions", "context", "instruction_format", "lang", "additional_rules"]
7
+
8
+ template = """
9
+ You are a highly intelligent language model trained to assist with a variety of language tasks. Your task here is to generate {num_questions} diverse questions or instructions based on the context provided below:
10
+
11
+ Context:
12
+ {context}
13
+
14
+ Please follow these rules:
15
+ {additional_rules}
16
+
17
+ Please generate the instructions in the {instruction_format} format and in {lang} language. Remember to adhere to the rules mentioned above.
18
+ """
19
+ template_format = "f-string"
20
+ def format(self, **kwargs):
21
+ """Format the prompt."""
22
+ return self.template.format(**kwargs)
23
+
requirements.txt ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ streamlit
2
+ pdfplumber
3
+ docx2txt
4
+ justext