Spaces:
Sleeping
Sleeping
Initial Commit
Browse files- app.py +302 -0
- custom_prompt_template.py +23 -0
- requirements.txt +4 -0
app.py
ADDED
|
@@ -0,0 +1,302 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import streamlit as st
|
| 2 |
+
import requests
|
| 3 |
+
import justext
|
| 4 |
+
import pdfplumber
|
| 5 |
+
import docx2txt
|
| 6 |
+
import test_prompt as prompt
|
| 7 |
+
import json
|
| 8 |
+
import os
|
| 9 |
+
import openai
|
| 10 |
+
|
| 11 |
+
from custom_prompt_template import InstructionGenerationTemplate
|
| 12 |
+
|
| 13 |
+
|
| 14 |
+
st.set_page_config(page_title="LLM instruction")
|
| 15 |
+
|
| 16 |
+
st.sidebar.success("Select a page above")
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
# function for the odia stoplists justext
|
| 20 |
+
def odia_stoplist():
|
| 21 |
+
odia_stopwords = [
|
| 22 |
+
"ଏହି", "ଏକ", "ଏକାଉଣଟ", "ମୁଁ", "ମୋର", "ମୁଁ ନିଜେ", "ଆମେ", "ଆମର", "ଆମର", "ଆମେ ନିଜେ", "ତୁମେ", "ତୁମର", "ତୁମର",
|
| 23 |
+
"ନିଜେ", "ନିଜେ", "ସେ", "ତାଙ୍କୁ", "ତାଙ୍କର",
|
| 24 |
+
"ନିଜେ", "ସେ", "ତାଙ୍କୁ", "ତାଙ୍କର", "ନିଜେ", "ଏହା", "ଏହାର", "ନିଜେ |", "ସେମାନେ", "ସେଗୁଡିକ", "ସେମାନଙ୍କର",
|
| 25 |
+
"ସେମାନଙ୍କର", "ନିଜେ |", "କଣ", "ଯାହା", "କିଏ", "କାହାକୁ",
|
| 26 |
+
"ଏହା", "ତାହା", "ଏଗୁଡ଼ିକ", "ସେଗୁଡ଼ିକ", "ମୁଁ", "ହେଉଛି", "ହେଉଛି |", "ଥିଲା", "ଥିଲା |", "ହୁଅ", "ହୋଇସାରିଛି |", "ହେବା",
|
| 27 |
+
"ଅଛି", "ଅଛି", "ଥିଲା", "ଅଛି", "କର", "କରେ |",
|
| 28 |
+
"କରିଛନ୍ତି", "କରିବା", "ଏବଂ", "କିନ୍ତୁ", "ଯଦି", "କିମ୍ବା", "କାରଣ", "ଯେପରି", "ପର୍ଯ୍ୟନ୍ତ", "ଯେତେବେଳେ", "ର", "ପାଇଁ",
|
| 29 |
+
"ସହିତ", "ବିଷୟରେ", "ବିପକ୍ଷରେ", "ମଧ୍ୟରେ", "ଭିତରକୁ", "ମାଧ୍ୟମରେ",
|
| 30 |
+
"ସମୟରେ", "ପୂର୍ବରୁ", "ପରେ", "ଉପରେ", "ନିମ୍ନରେ |", "କୁ", "ଠାରୁ", "ଅପ୍", "ତଳକୁ", "ଭିତରେ", "ବାହାରେ", "ଉପରେ", "ବନ୍ଦ",
|
| 31 |
+
"ସମାପ୍ତ", "ତଳେ |", "ପୁନର୍ବାର", "ଆଗକୁ",
|
| 32 |
+
"ତାପରେ", "ଥରେ |", "ଏଠାରେ", "ସେଠାରେ", "କେବେ", "କେଉଁଠାରେ", "କିପରି", "ସମସ୍ତ", "ଉଭୟ", "ପ୍ରତ୍ୟେକ", "ଅଳ୍ପ", "ଅଧିକ",
|
| 33 |
+
"ଅଧିକାଂଶ", "ଅନ୍ୟ", "କେତେକ", "ଏହିପରି",
|
| 34 |
+
"ନୁହେଁ |", "କେବଳ", "ନିଜର", "ସମାନ", "ତେଣୁ", "ଅପେକ୍ଷା", "ମଧ୍ୟ", "ବହୁତ", "କରିପାରିବେ |", "ଇଚ୍ଛା", "କେବଳ",
|
| 35 |
+
"କରିବା ଉଚିତ", "ବର୍ତ୍ତମାନ"
|
| 36 |
+
]
|
| 37 |
+
return frozenset(odia_stopwords)
|
| 38 |
+
|
| 39 |
+
|
| 40 |
+
# function to extract data from url using justext
|
| 41 |
+
def extract_data_from_url(url, language):
|
| 42 |
+
try:
|
| 43 |
+
response = requests.get(url)
|
| 44 |
+
response.raise_for_status()
|
| 45 |
+
page = response.content
|
| 46 |
+
|
| 47 |
+
para = ""
|
| 48 |
+
if language == "English":
|
| 49 |
+
paragraphs = justext.justext(page, justext.get_stoplist("English"))
|
| 50 |
+
elif language == "Hindi":
|
| 51 |
+
paragraphs = justext.justext(page, justext.get_stoplist("Hindi"))
|
| 52 |
+
elif language == "Odia":
|
| 53 |
+
paragraphs = justext.justext(
|
| 54 |
+
page, odia_stoplist(), 70, 140, 0.0, 0.02, 0.5, 150, False
|
| 55 |
+
)
|
| 56 |
+
|
| 57 |
+
for paragraph in paragraphs:
|
| 58 |
+
if not paragraph.is_boilerplate:
|
| 59 |
+
para = para + "\n" + paragraph.text
|
| 60 |
+
# returning the extracted data i.e para as string
|
| 61 |
+
return para
|
| 62 |
+
except Exception as e:
|
| 63 |
+
st.error(e)
|
| 64 |
+
|
| 65 |
+
|
| 66 |
+
# function to extract data from documents
|
| 67 |
+
def extract_data_from_documents(documents):
|
| 68 |
+
data = ""
|
| 69 |
+
if documents is not None:
|
| 70 |
+
for document in documents:
|
| 71 |
+
document_details = {
|
| 72 |
+
"filename": document.name,
|
| 73 |
+
"filetype": document.type,
|
| 74 |
+
"filesize": document.size,
|
| 75 |
+
}
|
| 76 |
+
st.write(document_details)
|
| 77 |
+
|
| 78 |
+
# Extract content from the txt file
|
| 79 |
+
if document.type == "text/plain":
|
| 80 |
+
# Read as bytes
|
| 81 |
+
data += str(document.read(), "utf-8")
|
| 82 |
+
|
| 83 |
+
# Extract content from the pdf file
|
| 84 |
+
elif document.type == "application/pdf":
|
| 85 |
+
# using pdfplumber
|
| 86 |
+
try:
|
| 87 |
+
with pdfplumber.open(document) as pdf:
|
| 88 |
+
all_text = ""
|
| 89 |
+
for page in pdf.pages:
|
| 90 |
+
text = page.extract_text()
|
| 91 |
+
all_text += text + "\n"
|
| 92 |
+
data += all_text
|
| 93 |
+
except requests.exceptions.RequestException as e:
|
| 94 |
+
st.write("None")
|
| 95 |
+
|
| 96 |
+
# Extract content from the docx file
|
| 97 |
+
elif (
|
| 98 |
+
document.type
|
| 99 |
+
== "application/vnd.openxmlformats-officedocument.wordprocessingml.document"
|
| 100 |
+
):
|
| 101 |
+
data += docx2txt.process(document)
|
| 102 |
+
|
| 103 |
+
# return extract data
|
| 104 |
+
return data
|
| 105 |
+
else:
|
| 106 |
+
st.error("Error: An error occurred while fetching content.")
|
| 107 |
+
# return extract status, and the data extracted
|
| 108 |
+
return None
|
| 109 |
+
|
| 110 |
+
|
| 111 |
+
# function for the keyboard
|
| 112 |
+
|
| 113 |
+
|
| 114 |
+
|
| 115 |
+
# Check the inputs for language, promptType
|
| 116 |
+
def valid_drop_down(language, promptType, noOfQuestions, instructionFormat):
|
| 117 |
+
langFlag = False
|
| 118 |
+
promptFlag = False
|
| 119 |
+
noOfQuestionFlag = False
|
| 120 |
+
instructionFormatFlag = False
|
| 121 |
+
|
| 122 |
+
if language:
|
| 123 |
+
langFlag = True
|
| 124 |
+
if promptType:
|
| 125 |
+
promptFlag = True
|
| 126 |
+
if noOfQuestions:
|
| 127 |
+
noOfQuestionFlag = True
|
| 128 |
+
if instructionFormat:
|
| 129 |
+
instructionFormatFlag = True
|
| 130 |
+
# checking for the compalsory inputs and return true only if all are set
|
| 131 |
+
return langFlag & promptFlag & noOfQuestionFlag & instructionFormatFlag
|
| 132 |
+
|
| 133 |
+
|
| 134 |
+
def main():
|
| 135 |
+
# setting up the initial session_states
|
| 136 |
+
if "extract_button" not in st.session_state:
|
| 137 |
+
st.session_state.extract_button = False
|
| 138 |
+
if "submit" not in st.session_state:
|
| 139 |
+
st.session_state.submit = False
|
| 140 |
+
if "generated" not in st.session_state:
|
| 141 |
+
st.session_state.generated = False
|
| 142 |
+
|
| 143 |
+
st.subheader("LLM Instructions")
|
| 144 |
+
|
| 145 |
+
# form to get the inputs
|
| 146 |
+
with st.form(key="form1"):
|
| 147 |
+
st.write("#")
|
| 148 |
+
|
| 149 |
+
# dropdown for language
|
| 150 |
+
language = st.selectbox("Select a language", ("", "English", "Hindi", "Odia"))
|
| 151 |
+
|
| 152 |
+
# dropdown for prompt type
|
| 153 |
+
promptType = st.selectbox(
|
| 154 |
+
"Select the Prompt type", ("", "Input text", "Url", "Document")
|
| 155 |
+
)
|
| 156 |
+
# inputs for number
|
| 157 |
+
noOfQuestions = st.number_input(
|
| 158 |
+
"Number of questions to generate:", min_value=1, max_value=20, value=10
|
| 159 |
+
)
|
| 160 |
+
|
| 161 |
+
# dropdown for language
|
| 162 |
+
instructionFormat = st.selectbox(
|
| 163 |
+
"Format of instruction:", ("Imperative sentence", "Question")
|
| 164 |
+
)
|
| 165 |
+
|
| 166 |
+
# checkbox for additional info bool val
|
| 167 |
+
addInfoCheckbox = st.checkbox("Input Additional Instructions", value=False)
|
| 168 |
+
|
| 169 |
+
st.write("##")
|
| 170 |
+
|
| 171 |
+
# form submit button and setting up the session_state
|
| 172 |
+
if st.form_submit_button():
|
| 173 |
+
st.session_state.submit = True
|
| 174 |
+
|
| 175 |
+
if st.session_state.submit:
|
| 176 |
+
# extends the prompt form to extract the data
|
| 177 |
+
with st.expander(label="prompt"):
|
| 178 |
+
with st.form(key="form2"):
|
| 179 |
+
# calling the function inside if to check valid drop down inputs
|
| 180 |
+
if valid_drop_down(
|
| 181 |
+
language, promptType, noOfQuestions, instructionFormat
|
| 182 |
+
):
|
| 183 |
+
if promptType == "Input text":
|
| 184 |
+
inputText = st.text_area(
|
| 185 |
+
label="For Instructions",
|
| 186 |
+
placeholder="Please enter your text here",
|
| 187 |
+
)
|
| 188 |
+
|
| 189 |
+
elif promptType == "Url":
|
| 190 |
+
url = st.text_input(
|
| 191 |
+
label="For URL", placeholder="Please enter your text here"
|
| 192 |
+
)
|
| 193 |
+
elif promptType == "Document":
|
| 194 |
+
documents = st.file_uploader(
|
| 195 |
+
label="For Documents ( pdf / txt / docx )",
|
| 196 |
+
type=["pdf", "txt", "docx"],
|
| 197 |
+
accept_multiple_files=True,
|
| 198 |
+
)
|
| 199 |
+
|
| 200 |
+
if addInfoCheckbox:
|
| 201 |
+
additionalInfo = st.text_input(
|
| 202 |
+
label="Additional Instructions",
|
| 203 |
+
placeholder="Please enter your text here",
|
| 204 |
+
)
|
| 205 |
+
|
| 206 |
+
if st.form_submit_button():
|
| 207 |
+
st.session_state.extract_button = True
|
| 208 |
+
# st.experimental_rerun()
|
| 209 |
+
|
| 210 |
+
# extracting data
|
| 211 |
+
if st.session_state.extract_button:
|
| 212 |
+
# extracting data
|
| 213 |
+
if promptType == "Input text":
|
| 214 |
+
extractedData = inputText
|
| 215 |
+
|
| 216 |
+
elif promptType == "Url":
|
| 217 |
+
extractedURLData = extract_data_from_url(url, language)
|
| 218 |
+
extractedData = extractedURLData
|
| 219 |
+
|
| 220 |
+
elif promptType == "Document":
|
| 221 |
+
if not documents:
|
| 222 |
+
documents = None
|
| 223 |
+
else:
|
| 224 |
+
for doc in documents:
|
| 225 |
+
if doc.name.split(".")[-1].lower() not in ["pdf", "txt", "docx"]:
|
| 226 |
+
# if documents is not the relevant type
|
| 227 |
+
st.error("Unsupported file: " + doc.name)
|
| 228 |
+
|
| 229 |
+
extractedDocumentData = extract_data_from_documents(documents)
|
| 230 |
+
extractedData = extractedDocumentData
|
| 231 |
+
|
| 232 |
+
# if the values are extracted running the custom prompt by creating an instance
|
| 233 |
+
if extractedData:
|
| 234 |
+
# ----------------------------- RUNNING THE PROMPT -----------------------------
|
| 235 |
+
|
| 236 |
+
# running the prompt form here
|
| 237 |
+
openai.api_key = os.getenv("OPENAI_API_KEY")
|
| 238 |
+
my_prompt_template = InstructionGenerationTemplate()
|
| 239 |
+
|
| 240 |
+
# providing the rules for the instructions to be generated
|
| 241 |
+
additional_rules = """
|
| 242 |
+
- You do not need to provide a response to the generated examples.
|
| 243 |
+
- You must return the response in the specified language.
|
| 244 |
+
- Each generated instruction can be either an imperative sentence or a question.
|
| 245 |
+
- Return the result in dictionary , where the key is the serial number and the value is an instruction.
|
| 246 |
+
"""
|
| 247 |
+
|
| 248 |
+
if st.button("Generate Instructions"):
|
| 249 |
+
prompt = my_prompt_template.format(
|
| 250 |
+
num_questions=noOfQuestions,
|
| 251 |
+
context=extractedData,
|
| 252 |
+
instruction_format=instructionFormat,
|
| 253 |
+
lang=language,
|
| 254 |
+
additional_rules=additional_rules
|
| 255 |
+
)
|
| 256 |
+
response = openai.ChatCompletion.create(
|
| 257 |
+
model="gpt-3.5-turbo",
|
| 258 |
+
messages=[
|
| 259 |
+
{"role": "system", "content": prompt},
|
| 260 |
+
# {"role": "user", "content": f"Generate {num_questions} diverse questions based on the context provided below in the form of {instruction_format}.\n\n{context}\n\n{additional_rules}"},
|
| 261 |
+
])
|
| 262 |
+
|
| 263 |
+
if "result" not in st.session_state:
|
| 264 |
+
st.session_state["result"] = response.choices[0].message.content
|
| 265 |
+
st.session_state.generated = True
|
| 266 |
+
|
| 267 |
+
if st.session_state.generated:
|
| 268 |
+
# displaying the generated instructions
|
| 269 |
+
st.write("Generated Insuctions")
|
| 270 |
+
# st.write(result)
|
| 271 |
+
result = st.session_state["result"]
|
| 272 |
+
print(type(result))
|
| 273 |
+
print(result)
|
| 274 |
+
# cleaned_result=result.replace("<example>","").replace("</example>","").replace("\\n","\n")
|
| 275 |
+
result_dict=json.loads(result)
|
| 276 |
+
print(type(result_dict))
|
| 277 |
+
print(result_dict)
|
| 278 |
+
# print(result_dict)
|
| 279 |
+
# for key,value in result_dict.items():
|
| 280 |
+
# print(f"{key}:{value}")
|
| 281 |
+
# including the questions as checkboxes for generating further solutions
|
| 282 |
+
# Creating list to display the selected instructions
|
| 283 |
+
selected_items = [f"{value} " for key, value in result_dict.items() if st.checkbox(f"Q{key} : {value}")]
|
| 284 |
+
|
| 285 |
+
# Display the selected items as a list
|
| 286 |
+
if selected_items:
|
| 287 |
+
st.write("Selected Items:")
|
| 288 |
+
st.write(selected_items)
|
| 289 |
+
else:
|
| 290 |
+
st.write("No items selected.")
|
| 291 |
+
|
| 292 |
+
|
| 293 |
+
if st.button("clear"):
|
| 294 |
+
st.session_state.extract_button = False
|
| 295 |
+
st.session_state.submit = False
|
| 296 |
+
st.session_state.generated = True
|
| 297 |
+
del st.session_state["result"]
|
| 298 |
+
st.experimental_rerun()
|
| 299 |
+
|
| 300 |
+
|
| 301 |
+
if __name__ == "__main__":
|
| 302 |
+
main()
|
custom_prompt_template.py
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from typing import List
|
| 2 |
+
import langchain
|
| 3 |
+
class InstructionGenerationTemplate(langchain.prompts.PromptTemplate):
|
| 4 |
+
"""A custom prompt template for generating instructions."""
|
| 5 |
+
|
| 6 |
+
input_variables: List[str] = ["num_questions", "context", "instruction_format", "lang", "additional_rules"]
|
| 7 |
+
|
| 8 |
+
template = """
|
| 9 |
+
You are a highly intelligent language model trained to assist with a variety of language tasks. Your task here is to generate {num_questions} diverse questions or instructions based on the context provided below:
|
| 10 |
+
|
| 11 |
+
Context:
|
| 12 |
+
{context}
|
| 13 |
+
|
| 14 |
+
Please follow these rules:
|
| 15 |
+
{additional_rules}
|
| 16 |
+
|
| 17 |
+
Please generate the instructions in the {instruction_format} format and in {lang} language. Remember to adhere to the rules mentioned above.
|
| 18 |
+
"""
|
| 19 |
+
template_format = "f-string"
|
| 20 |
+
def format(self, **kwargs):
|
| 21 |
+
"""Format the prompt."""
|
| 22 |
+
return self.template.format(**kwargs)
|
| 23 |
+
|
requirements.txt
ADDED
|
@@ -0,0 +1,4 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
streamlit
|
| 2 |
+
pdfplumber
|
| 3 |
+
docx2txt
|
| 4 |
+
justext
|