kawaa99 commited on
Commit
a83ad3d
·
verified ·
1 Parent(s): 450e6c7

Update utils.py

Browse files
Files changed (1) hide show
  1. utils.py +18 -5
utils.py CHANGED
@@ -12,6 +12,18 @@ from functools import lru_cache
12
  from transformers import pipeline
13
  import re
14
 
 
 
 
 
 
 
 
 
 
 
 
 
15
  # LOAD IDIOMS
16
  @st.cache_data
17
  def load_idioms(path="idiom2.json"):
@@ -22,7 +34,7 @@ def load_idioms(path="idiom2.json"):
22
 
23
  for idiom, info in data.items():
24
 
25
- cleaned[idiom.lower()] = {
26
  "meaning": info.get("meaning", ""),
27
  "topic": info.get("topic", "General")
28
  }
@@ -148,7 +160,8 @@ def build_examples_map():
148
  examples_map = {}
149
 
150
  def add_example(key, en, es):
151
- key = key.lower().strip()
 
152
  if key not in examples_map:
153
  examples_map[key] = []
154
  if en or es:
@@ -339,8 +352,8 @@ def generate_ai_sentence(idiom, examples_map, used_structures):
339
  continue
340
 
341
  # ---------- FALLBACK TO DATASET ----------
342
- examples = examples_map.get(idiom.lower(), [])
343
-
344
  valid_examples = []
345
 
346
  for ex in examples:
@@ -411,7 +424,7 @@ def generate_adaptive_quiz(
411
  # ---------- DATASET FIRST (fast) ----------
412
  t0 = time.time()
413
  for idiom in available:
414
- examples = examples_map.get(idiom.lower(), [])
415
  for ex in examples:
416
  sentence = ex.get("en", "")
417
  if not sentence:
 
12
  from transformers import pipeline
13
  import re
14
 
15
+ def normalize_idiom(text):
16
+ """
17
+ Normalize an idiom string for consistent dictionary keys / matching:
18
+ - lowercase
19
+ - strip leading/trailing whitespace
20
+ - strip a leading "to " (dataset idioms sometimes include the infinitive marker)
21
+ """
22
+ text = text.strip().lower()
23
+ if text.startswith("to "):
24
+ text = text[3:].strip()
25
+ return text
26
+
27
  # LOAD IDIOMS
28
  @st.cache_data
29
  def load_idioms(path="idiom2.json"):
 
34
 
35
  for idiom, info in data.items():
36
 
37
+ cleaned[normalize_idiom(idiom)] = {
38
  "meaning": info.get("meaning", ""),
39
  "topic": info.get("topic", "General")
40
  }
 
160
  examples_map = {}
161
 
162
  def add_example(key, en, es):
163
+ #key = key.lower().strip()
164
+ key = normalize_idiom(key)
165
  if key not in examples_map:
166
  examples_map[key] = []
167
  if en or es:
 
352
  continue
353
 
354
  # ---------- FALLBACK TO DATASET ----------
355
+ #examples = examples_map.get(idiom.lower(), [])
356
+ examples = examples_map.get(normalize_idiom(idiom), [])
357
  valid_examples = []
358
 
359
  for ex in examples:
 
424
  # ---------- DATASET FIRST (fast) ----------
425
  t0 = time.time()
426
  for idiom in available:
427
+ examples = examples_map.get(normalize_idiom(idiom), []) #changed
428
  for ex in examples:
429
  sentence = ex.get("en", "")
430
  if not sentence: