AIEcosystem commited on
Commit
e07548f
·
verified ·
1 Parent(s): fe14884

Update src/streamlit_app.py

Browse files
Files changed (1) hide show
  1. src/streamlit_app.py +12 -130
src/streamlit_app.py CHANGED
@@ -12,9 +12,6 @@ from streamlit_extras.stylable_container import stylable_container
12
  from typing import Optional
13
  from gliner import GLiNER
14
  from comet_ml import Experiment
15
-
16
-
17
-
18
  st.markdown(
19
  """
20
  <style>
@@ -23,24 +20,20 @@ st.markdown(
23
  background-color: #F5F5F5; /* A very light grey */
24
  color: #333333; /* Dark grey for text for good contrast */
25
  }
26
-
27
  /* Sidebar background */
28
  .css-1d36184, .css-1d36184, .st-ck {
29
  background-color: #D3D3D3; /* Light grey for the sidebar */
30
  }
31
-
32
  /* Expander header and content background */
33
  .streamlit-expanderHeader, .streamlit-expanderContent {
34
  background-color: #F5F5F5;
35
  }
36
-
37
  /* Text Area background and text color */
38
  .stTextArea textarea {
39
  background-color: #E6E6E6; /* Slightly darker grey for input fields */
40
  color: #000000;
41
  border: 1px solid #B0B0B0; /* Add a subtle border */
42
  }
43
-
44
  /* Button styling */
45
  .stButton > button {
46
  background-color: #B0B0B0; /* A medium grey for the button */
@@ -52,7 +45,6 @@ st.markdown(
52
  .stButton > button:hover {
53
  background-color: #8C8C8C; /* Darker grey on hover */
54
  }
55
-
56
  /* Alert boxes */
57
  .stAlert {
58
  color: #000000;
@@ -66,42 +58,18 @@ st.markdown(
66
  }
67
  </style>
68
  """,
69
- unsafe_allow_html=True
70
- )
71
-
72
-
73
-
74
-
75
-
76
  # --- Page Configuration and UI Elements ---
77
  st.set_page_config(layout="wide", page_title="Named Entity Recognition App")
78
  st.subheader("Legal Lens", divider="grey")
79
  st.link_button("by nlpblogs", "https://nlpblogs.com", type="tertiary")
80
  expander = st.expander("**Important notes**")
81
- expander.write("""**Named Entities:** This Legal Lens web app predicts twenty-eight (28) labels: "Plaintiff", "Defendant", "Appellant", "Appellee", "Debtor", "Creditor", "Signer", "Witness", "Courts", "Judges", "Lawyers", "Attorneys", "Statutes", "Laws", "Provisions", "Case_citations", "Legal_documents", "Effective_dates", "Execution_dates", "Expiration_dates", "Money", "Amounts", "Contract_terms", "Case_number", "Witnesses", "Crimes", "Offenses", "Victims"
82
-
83
- Results are presented in easy-to-read tables, visualized in an interactive tree map, pie chart and bar chart, and are available for download along with a Glossary of tags.
84
-
85
- **How to Use:** Type or paste your text into the text area below, then press Ctrl + Enter. Click the 'Results' button to extract and tag entities in your text data.
86
-
87
- **Usage Limits:** You can request results unlimited times for one (1) month.
88
-
89
- **Supported Languages:** English
90
-
91
- **Technical issues:** If your connection times out, please refresh the page or reopen the app's URL.
92
-
93
- For any errors or inquiries, please contact us at info@nlpblogs.com""")
94
-
95
  with st.sidebar:
96
  st.write("Use the following code to embed the Legal Lens web app on your website. Feel free to adjust the width and height values to fit your page.")
97
  code = '''
98
- <iframe
99
- src="https://aiecosystem-legal-lens.hf.space"
100
- frameborder="0"
101
- width="850"
102
- height="450"
103
  ></iframe>
104
-
105
  '''
106
  st.code(code, language="html")
107
  st.text("")
@@ -109,57 +77,19 @@ with st.sidebar:
109
  st.divider()
110
  st.subheader("🚀 Ready to build your own AI Web App?", divider="grey")
111
  st.link_button("AI Web App Builder", "https://nlpblogs.com/build-your-named-entity-recognition-app/", type="primary")
112
-
113
  # --- Comet ML Setup ---
114
  COMET_API_KEY = os.environ.get("COMET_API_KEY")
115
  COMET_WORKSPACE = os.environ.get("COMET_WORKSPACE")
116
  COMET_PROJECT_NAME = os.environ.get("COMET_PROJECT_NAME")
117
  comet_initialized = bool(COMET_API_KEY and COMET_WORKSPACE and COMET_PROJECT_NAME)
118
-
119
  if not comet_initialized:
120
  st.warning("Comet ML not initialized. Check environment variables.")
121
-
122
  # --- Label Definitions ---
123
-
124
- labels = [
125
- "Plaintiff",
126
- "Defendant",
127
- "Appellant",
128
- "Appellee",
129
- "Debtor",
130
- "Creditor",
131
- "Signer",
132
- "Witness",
133
- "Courts",
134
- "Judges",
135
- "Lawyers",
136
- "Attorneys",
137
- "Statutes",
138
- "Laws",
139
- "Provisions",
140
- "Case_citations",
141
- "Legal_documents",
142
- "Effective_dates",
143
- "Execution_dates",
144
- "Expiration_dates",
145
- "Money",
146
- "Amounts",
147
- "Contract_terms",
148
- "Case_number",
149
- "Witnesses",
150
- "Crimes",
151
- "Offenses",
152
- "Victims"
153
- ]
154
-
155
-
156
-
157
  category_mapping = {
158
  "Parties": [
159
  "Plaintiff",
160
-
161
  "Defendant",
162
-
163
  "Appellant",
164
  "Appellee",
165
  "Debtor",
@@ -167,67 +97,41 @@ category_mapping = {
167
  "Signer",
168
  "Witness"
169
  ],
170
-
171
  "Judicial & Governmental Bodies": [
172
  "Courts",
173
  "Judges",
174
  "Lawyers",
175
  "Attorneys"
176
-
177
  ],
178
-
179
  "Legal Instruments & Concepts": [
180
  "Statutes",
181
  "Laws",
182
  "Provisions",
183
  "Case_citations",
184
  "Legal_documents"
185
-
186
  ],
187
-
188
  "Dates & Timeframes": [
189
  "Effective_dates",
190
  "Execution_dates",
191
  "Expiration_dates"
192
-
193
-
194
  ],
195
-
196
  "Financial & Monetary Entities": [
197
  "Money",
198
  "Amounts"
199
-
200
  ],
201
-
202
  "Contracts": [
203
  "Contract_terms"
204
-
205
  ],
206
-
207
  "Court Judgments": [
208
  "Case_number",
209
  "Witnesses",
210
-
211
  ],
212
-
213
-
214
-
215
- "Criminal Law": [
216
  "Crimes",
217
  "Offenses",
218
  "Victims"
219
-
220
  ]
221
-
222
-
223
  }
224
-
225
-
226
-
227
-
228
-
229
-
230
-
231
  # --- Model Loading ---
232
  @st.cache_resource
233
  def load_ner_model():
@@ -238,30 +142,28 @@ def load_ner_model():
238
  st.error(f"Failed to load NER model. Please check your internet connection or model availability: {e}")
239
  st.stop()
240
  model = load_ner_model()
241
-
242
  # Flatten the mapping to a single dictionary
243
  reverse_category_mapping = {label: category for category, label_list in category_mapping.items() for label in label_list}
244
-
245
  # --- Text Input and Clear Button ---
246
- text = st.text_area("Type or paste your text below, and then press Ctrl + Enter", height=250, key='my_text_area')
247
-
 
 
248
  def clear_text():
249
  """Clears the text area."""
250
  st.session_state['my_text_area'] = ""
251
-
252
  st.button("Clear text", on_click=clear_text)
253
-
254
-
255
  # --- Results Section ---
256
  if st.button("Results"):
257
  start_time = time.time()
258
  if not text.strip():
259
  st.warning("Please enter some text to extract entities.")
 
 
260
  else:
261
  with st.spinner("Extracting entities...", show_time=True):
262
  entities = model.predict_entities(text, labels)
263
  df = pd.DataFrame(entities)
264
-
265
  if not df.empty:
266
  df['category'] = df['label'].map(reverse_category_mapping)
267
  if comet_initialized:
@@ -272,13 +174,10 @@ if st.button("Results"):
272
  )
273
  experiment.log_parameter("input_text", text)
274
  experiment.log_table("predicted_entities", df)
275
-
276
  st.subheader("Grouped Entities by Category", divider = "grey")
277
-
278
  # Create tabs for each category
279
  category_names = sorted(list(category_mapping.keys()))
280
  category_tabs = st.tabs(category_names)
281
-
282
  for i, category_name in enumerate(category_names):
283
  with category_tabs[i]:
284
  df_category_filtered = df[df['category'] == category_name]
@@ -286,9 +185,6 @@ if st.button("Results"):
286
  st.dataframe(df_category_filtered.drop(columns=['category']), use_container_width=True)
287
  else:
288
  st.info(f"No entities found for the '{category_name}' category.")
289
-
290
-
291
-
292
  with st.expander("See Glossary of tags"):
293
  st.write('''
294
  - **text**: ['entity extracted from your text data']
@@ -298,18 +194,15 @@ if st.button("Results"):
298
  - **end**: ['index of the end of the corresponding entity']
299
  ''')
300
  st.divider()
301
-
302
  # Tree map
303
  st.subheader("Tree map", divider = "grey")
304
  fig_treemap = px.treemap(df, path=[px.Constant("all"), 'category', 'label', 'text'], values='score', color='category')
305
  fig_treemap.update_layout(margin=dict(t=50, l=25, r=25, b=25), paper_bgcolor='#F5F5F5', plot_bgcolor='#F5F5F5')
306
  st.plotly_chart(fig_treemap)
307
-
308
  # Pie and Bar charts
309
  grouped_counts = df['category'].value_counts().reset_index()
310
  grouped_counts.columns = ['category', 'count']
311
  col1, col2 = st.columns(2)
312
-
313
  with col1:
314
  st.subheader("Pie chart", divider = "grey")
315
  fig_pie = px.pie(grouped_counts, values='count', names='category', hover_data=['count'], labels={'count': 'count'}, title='Percentage of predicted categories')
@@ -319,10 +212,6 @@ if st.button("Results"):
319
  plot_bgcolor='#F5F5F5'
320
  )
321
  st.plotly_chart(fig_pie)
322
-
323
-
324
-
325
-
326
  with col2:
327
  st.subheader("Bar chart", divider = "grey")
328
  fig_bar = px.bar(grouped_counts, x="count", y="category", color="category", text_auto=True, title='Occurrences of predicted categories')
@@ -331,7 +220,6 @@ if st.button("Results"):
331
  plot_bgcolor='#F5F5F5'
332
  )
333
  st.plotly_chart(fig_bar)
334
-
335
  # Most Frequent Entities
336
  st.subheader("Most Frequent Entities", divider="grey")
337
  word_counts = df['text'].value_counts().reset_index()
@@ -346,10 +234,8 @@ if st.button("Results"):
346
  st.plotly_chart(fig_repeating_bar)
347
  else:
348
  st.warning("No entities were found that occur more than once.")
349
-
350
  # Download Section
351
  st.divider()
352
-
353
  dfa = pd.DataFrame(
354
  data={
355
  'Column Name': ['text', 'label', 'score', 'start', 'end'],
@@ -359,7 +245,6 @@ if st.button("Results"):
359
  'accuracy score; how accurately a tag has been assigned to a given entity',
360
  'index of the start of the corresponding entity',
361
  'index of the end of the corresponding entity',
362
-
363
  ]
364
  }
365
  )
@@ -367,7 +252,6 @@ if st.button("Results"):
367
  with zipfile.ZipFile(buf, "w") as myzip:
368
  myzip.writestr("Summary of the results.csv", df.to_csv(index=False))
369
  myzip.writestr("Glossary of tags.csv", dfa.to_csv(index=False))
370
-
371
  with stylable_container(
372
  key="download_button",
373
  css_styles="""button { background-color: red; border: 1px solid black; padding: 5px; color: white; }""",
@@ -378,14 +262,12 @@ if st.button("Results"):
378
  file_name="nlpblogs_results.zip",
379
  mime="application/zip",
380
  )
381
-
382
  if comet_initialized:
383
  experiment.log_figure(figure=fig_treemap, figure_name="entity_treemap_categories")
384
  experiment.end()
385
  else: # If df is empty
386
  st.warning("No entities were found in the provided text.")
387
-
388
- end_time = time.time()
389
  elapsed_time = end_time - start_time
390
  st.text("")
391
  st.text("")
 
12
  from typing import Optional
13
  from gliner import GLiNER
14
  from comet_ml import Experiment
 
 
 
15
  st.markdown(
16
  """
17
  <style>
 
20
  background-color: #F5F5F5; /* A very light grey */
21
  color: #333333; /* Dark grey for text for good contrast */
22
  }
 
23
  /* Sidebar background */
24
  .css-1d36184, .css-1d36184, .st-ck {
25
  background-color: #D3D3D3; /* Light grey for the sidebar */
26
  }
 
27
  /* Expander header and content background */
28
  .streamlit-expanderHeader, .streamlit-expanderContent {
29
  background-color: #F5F5F5;
30
  }
 
31
  /* Text Area background and text color */
32
  .stTextArea textarea {
33
  background-color: #E6E6E6; /* Slightly darker grey for input fields */
34
  color: #000000;
35
  border: 1px solid #B0B0B0; /* Add a subtle border */
36
  }
 
37
  /* Button styling */
38
  .stButton > button {
39
  background-color: #B0B0B0; /* A medium grey for the button */
 
45
  .stButton > button:hover {
46
  background-color: #8C8C8C; /* Darker grey on hover */
47
  }
 
48
  /* Alert boxes */
49
  .stAlert {
50
  color: #000000;
 
58
  }
59
  </style>
60
  """,
61
+ unsafe_allow_html=True)
 
 
 
 
 
 
62
  # --- Page Configuration and UI Elements ---
63
  st.set_page_config(layout="wide", page_title="Named Entity Recognition App")
64
  st.subheader("Legal Lens", divider="grey")
65
  st.link_button("by nlpblogs", "https://nlpblogs.com", type="tertiary")
66
  expander = st.expander("**Important notes**")
67
+ expander.write("""**Named Entities:** This Legal Lens web app predicts twenty-eight (28) labels: "Plaintiff", "Defendant", "Appellant", "Appellee", "Debtor", "Creditor", "Signer", "Witness", "Courts", "Judges", "Lawyers", "Attorneys", "Statutes", "Laws", "Provisions", "Case_citations", "Legal_documents", "Effective_dates", "Execution_dates", "Expiration_dates", "Money", "Amounts", "Contract_terms", "Case_number", "Witnesses", "Crimes", "Offenses", "Victims"Results are presented in easy-to-read tables, visualized in an interactive tree map, pie chart and bar chart, and are available for download along with a Glossary of tags. **How to Use:** Type or paste your text into the text area below, then press Ctrl + Enter. Click the 'Results' button to extract and tag entities in your text data. **Usage Limits:** You can request results unlimited times for one (1) month. **Supported Languages:** English **Technical issues:** If your connection times out, please refresh the page or reopen the app's URL. For any errors or inquiries, please contact us at info@nlpblogs.com""")
 
 
 
 
 
 
 
 
 
 
 
 
 
68
  with st.sidebar:
69
  st.write("Use the following code to embed the Legal Lens web app on your website. Feel free to adjust the width and height values to fit your page.")
70
  code = '''
71
+ <iframe src="https://aiecosystem-legal-lens.hf.space" frameborder="0" width="850" height="450"
 
 
 
 
72
  ></iframe>
 
73
  '''
74
  st.code(code, language="html")
75
  st.text("")
 
77
  st.divider()
78
  st.subheader("🚀 Ready to build your own AI Web App?", divider="grey")
79
  st.link_button("AI Web App Builder", "https://nlpblogs.com/build-your-named-entity-recognition-app/", type="primary")
 
80
  # --- Comet ML Setup ---
81
  COMET_API_KEY = os.environ.get("COMET_API_KEY")
82
  COMET_WORKSPACE = os.environ.get("COMET_WORKSPACE")
83
  COMET_PROJECT_NAME = os.environ.get("COMET_PROJECT_NAME")
84
  comet_initialized = bool(COMET_API_KEY and COMET_WORKSPACE and COMET_PROJECT_NAME)
 
85
  if not comet_initialized:
86
  st.warning("Comet ML not initialized. Check environment variables.")
 
87
  # --- Label Definitions ---
88
+ labels = ["Plaintiff","Defendant","Appellant","Appellee","Debtor","Creditor","Signer","Witness","Courts","Judges","Lawyers","Attorneys","Statutes","Laws","Provisions","Case_citations","Legal_documents","Effective_dates","Execution_dates","Expiration_dates","Money","Amounts","Contract_terms","Case_number","Witnesses","Crimes","Offenses","Victims"]
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
89
  category_mapping = {
90
  "Parties": [
91
  "Plaintiff",
 
92
  "Defendant",
 
93
  "Appellant",
94
  "Appellee",
95
  "Debtor",
 
97
  "Signer",
98
  "Witness"
99
  ],
 
100
  "Judicial & Governmental Bodies": [
101
  "Courts",
102
  "Judges",
103
  "Lawyers",
104
  "Attorneys"
 
105
  ],
 
106
  "Legal Instruments & Concepts": [
107
  "Statutes",
108
  "Laws",
109
  "Provisions",
110
  "Case_citations",
111
  "Legal_documents"
 
112
  ],
 
113
  "Dates & Timeframes": [
114
  "Effective_dates",
115
  "Execution_dates",
116
  "Expiration_dates"
 
 
117
  ],
 
118
  "Financial & Monetary Entities": [
119
  "Money",
120
  "Amounts"
 
121
  ],
 
122
  "Contracts": [
123
  "Contract_terms"
 
124
  ],
 
125
  "Court Judgments": [
126
  "Case_number",
127
  "Witnesses",
 
128
  ],
129
+ "Criminal Law": [
 
 
 
130
  "Crimes",
131
  "Offenses",
132
  "Victims"
 
133
  ]
 
 
134
  }
 
 
 
 
 
 
 
135
  # --- Model Loading ---
136
  @st.cache_resource
137
  def load_ner_model():
 
142
  st.error(f"Failed to load NER model. Please check your internet connection or model availability: {e}")
143
  st.stop()
144
  model = load_ner_model()
 
145
  # Flatten the mapping to a single dictionary
146
  reverse_category_mapping = {label: category for category, label_list in category_mapping.items() for label in label_list}
 
147
  # --- Text Input and Clear Button ---
148
+ word_limit = 200
149
+ text = st.text_area(f"Type or paste your text below (max {word_limit} words), and then press Ctrl + Enter", height=250, key='my_text_area')
150
+ word_count = len(text.split())
151
+ st.markdown(f"**Word count:** {word_count}/{word_limit}")
152
  def clear_text():
153
  """Clears the text area."""
154
  st.session_state['my_text_area'] = ""
 
155
  st.button("Clear text", on_click=clear_text)
 
 
156
  # --- Results Section ---
157
  if st.button("Results"):
158
  start_time = time.time()
159
  if not text.strip():
160
  st.warning("Please enter some text to extract entities.")
161
+ elif word_count > word_limit:
162
+ st.warning(f"Your text exceeds the {word_limit} word limit. Please shorten it to continue.")
163
  else:
164
  with st.spinner("Extracting entities...", show_time=True):
165
  entities = model.predict_entities(text, labels)
166
  df = pd.DataFrame(entities)
 
167
  if not df.empty:
168
  df['category'] = df['label'].map(reverse_category_mapping)
169
  if comet_initialized:
 
174
  )
175
  experiment.log_parameter("input_text", text)
176
  experiment.log_table("predicted_entities", df)
 
177
  st.subheader("Grouped Entities by Category", divider = "grey")
 
178
  # Create tabs for each category
179
  category_names = sorted(list(category_mapping.keys()))
180
  category_tabs = st.tabs(category_names)
 
181
  for i, category_name in enumerate(category_names):
182
  with category_tabs[i]:
183
  df_category_filtered = df[df['category'] == category_name]
 
185
  st.dataframe(df_category_filtered.drop(columns=['category']), use_container_width=True)
186
  else:
187
  st.info(f"No entities found for the '{category_name}' category.")
 
 
 
188
  with st.expander("See Glossary of tags"):
189
  st.write('''
190
  - **text**: ['entity extracted from your text data']
 
194
  - **end**: ['index of the end of the corresponding entity']
195
  ''')
196
  st.divider()
 
197
  # Tree map
198
  st.subheader("Tree map", divider = "grey")
199
  fig_treemap = px.treemap(df, path=[px.Constant("all"), 'category', 'label', 'text'], values='score', color='category')
200
  fig_treemap.update_layout(margin=dict(t=50, l=25, r=25, b=25), paper_bgcolor='#F5F5F5', plot_bgcolor='#F5F5F5')
201
  st.plotly_chart(fig_treemap)
 
202
  # Pie and Bar charts
203
  grouped_counts = df['category'].value_counts().reset_index()
204
  grouped_counts.columns = ['category', 'count']
205
  col1, col2 = st.columns(2)
 
206
  with col1:
207
  st.subheader("Pie chart", divider = "grey")
208
  fig_pie = px.pie(grouped_counts, values='count', names='category', hover_data=['count'], labels={'count': 'count'}, title='Percentage of predicted categories')
 
212
  plot_bgcolor='#F5F5F5'
213
  )
214
  st.plotly_chart(fig_pie)
 
 
 
 
215
  with col2:
216
  st.subheader("Bar chart", divider = "grey")
217
  fig_bar = px.bar(grouped_counts, x="count", y="category", color="category", text_auto=True, title='Occurrences of predicted categories')
 
220
  plot_bgcolor='#F5F5F5'
221
  )
222
  st.plotly_chart(fig_bar)
 
223
  # Most Frequent Entities
224
  st.subheader("Most Frequent Entities", divider="grey")
225
  word_counts = df['text'].value_counts().reset_index()
 
234
  st.plotly_chart(fig_repeating_bar)
235
  else:
236
  st.warning("No entities were found that occur more than once.")
 
237
  # Download Section
238
  st.divider()
 
239
  dfa = pd.DataFrame(
240
  data={
241
  'Column Name': ['text', 'label', 'score', 'start', 'end'],
 
245
  'accuracy score; how accurately a tag has been assigned to a given entity',
246
  'index of the start of the corresponding entity',
247
  'index of the end of the corresponding entity',
 
248
  ]
249
  }
250
  )
 
252
  with zipfile.ZipFile(buf, "w") as myzip:
253
  myzip.writestr("Summary of the results.csv", df.to_csv(index=False))
254
  myzip.writestr("Glossary of tags.csv", dfa.to_csv(index=False))
 
255
  with stylable_container(
256
  key="download_button",
257
  css_styles="""button { background-color: red; border: 1px solid black; padding: 5px; color: white; }""",
 
262
  file_name="nlpblogs_results.zip",
263
  mime="application/zip",
264
  )
 
265
  if comet_initialized:
266
  experiment.log_figure(figure=fig_treemap, figure_name="entity_treemap_categories")
267
  experiment.end()
268
  else: # If df is empty
269
  st.warning("No entities were found in the provided text.")
270
+ end_time = time.time()
 
271
  elapsed_time = end_time - start_time
272
  st.text("")
273
  st.text("")