Spaces:
Sleeping
Sleeping
Update utils.py
Browse files
utils.py
CHANGED
|
@@ -12,6 +12,18 @@ from functools import lru_cache
|
|
| 12 |
from transformers import pipeline
|
| 13 |
import re
|
| 14 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 15 |
# LOAD IDIOMS
|
| 16 |
@st.cache_data
|
| 17 |
def load_idioms(path="idiom2.json"):
|
|
@@ -22,7 +34,7 @@ def load_idioms(path="idiom2.json"):
|
|
| 22 |
|
| 23 |
for idiom, info in data.items():
|
| 24 |
|
| 25 |
-
cleaned[
|
| 26 |
"meaning": info.get("meaning", ""),
|
| 27 |
"topic": info.get("topic", "General")
|
| 28 |
}
|
|
@@ -148,7 +160,8 @@ def build_examples_map():
|
|
| 148 |
examples_map = {}
|
| 149 |
|
| 150 |
def add_example(key, en, es):
|
| 151 |
-
key = key.lower().strip()
|
|
|
|
| 152 |
if key not in examples_map:
|
| 153 |
examples_map[key] = []
|
| 154 |
if en or es:
|
|
@@ -339,8 +352,8 @@ def generate_ai_sentence(idiom, examples_map, used_structures):
|
|
| 339 |
continue
|
| 340 |
|
| 341 |
# ---------- FALLBACK TO DATASET ----------
|
| 342 |
-
examples = examples_map.get(idiom.lower(), [])
|
| 343 |
-
|
| 344 |
valid_examples = []
|
| 345 |
|
| 346 |
for ex in examples:
|
|
@@ -411,7 +424,7 @@ def generate_adaptive_quiz(
|
|
| 411 |
# ---------- DATASET FIRST (fast) ----------
|
| 412 |
t0 = time.time()
|
| 413 |
for idiom in available:
|
| 414 |
-
examples = examples_map.get(
|
| 415 |
for ex in examples:
|
| 416 |
sentence = ex.get("en", "")
|
| 417 |
if not sentence:
|
|
|
|
| 12 |
from transformers import pipeline
|
| 13 |
import re
|
| 14 |
|
| 15 |
+
def normalize_idiom(text):
|
| 16 |
+
"""
|
| 17 |
+
Normalize an idiom string for consistent dictionary keys / matching:
|
| 18 |
+
- lowercase
|
| 19 |
+
- strip leading/trailing whitespace
|
| 20 |
+
- strip a leading "to " (dataset idioms sometimes include the infinitive marker)
|
| 21 |
+
"""
|
| 22 |
+
text = text.strip().lower()
|
| 23 |
+
if text.startswith("to "):
|
| 24 |
+
text = text[3:].strip()
|
| 25 |
+
return text
|
| 26 |
+
|
| 27 |
# LOAD IDIOMS
|
| 28 |
@st.cache_data
|
| 29 |
def load_idioms(path="idiom2.json"):
|
|
|
|
| 34 |
|
| 35 |
for idiom, info in data.items():
|
| 36 |
|
| 37 |
+
cleaned[normalize_idiom(idiom)] = {
|
| 38 |
"meaning": info.get("meaning", ""),
|
| 39 |
"topic": info.get("topic", "General")
|
| 40 |
}
|
|
|
|
| 160 |
examples_map = {}
|
| 161 |
|
| 162 |
def add_example(key, en, es):
|
| 163 |
+
#key = key.lower().strip()
|
| 164 |
+
key = normalize_idiom(key)
|
| 165 |
if key not in examples_map:
|
| 166 |
examples_map[key] = []
|
| 167 |
if en or es:
|
|
|
|
| 352 |
continue
|
| 353 |
|
| 354 |
# ---------- FALLBACK TO DATASET ----------
|
| 355 |
+
#examples = examples_map.get(idiom.lower(), [])
|
| 356 |
+
examples = examples_map.get(normalize_idiom(idiom), [])
|
| 357 |
valid_examples = []
|
| 358 |
|
| 359 |
for ex in examples:
|
|
|
|
| 424 |
# ---------- DATASET FIRST (fast) ----------
|
| 425 |
t0 = time.time()
|
| 426 |
for idiom in available:
|
| 427 |
+
examples = examples_map.get(normalize_idiom(idiom), []) #changed
|
| 428 |
for ex in examples:
|
| 429 |
sentence = ex.get("en", "")
|
| 430 |
if not sentence:
|