kawaa99 commited on
Commit
f3494cd
·
verified ·
1 Parent(s): 09415a1

Update utils.py

Browse files
Files changed (1) hide show
  1. utils.py +64 -20
utils.py CHANGED
@@ -200,7 +200,6 @@ def normalize_structure(sentence):
200
 
201
  return " ".join(words[:3])
202
 
203
-
204
  def generate_ai_sentence(idiom, examples_map, used_structures):
205
 
206
  subjects = [
@@ -223,36 +222,48 @@ def generate_ai_sentence(idiom, examples_map, used_structures):
223
 
224
  for _ in range(8):
225
 
226
- subject = random.choice(subjects)
227
- tone = random.choice(tones)
 
 
228
 
229
- prompt = f"""
230
- Sentence with the idiom "{idiom}":
231
- """
 
 
232
 
233
- try:
234
  result = generator(
235
  prompt,
236
- max_new_tokens=50,
237
  do_sample=True,
238
- temperature=0.8
 
 
 
 
239
  )
240
-
241
- sentence = result[0]["generated_text"].strip()
242
 
 
 
 
 
243
  if not sentence:
244
  continue
245
 
 
 
 
246
  if idiom.lower() not in sentence.lower():
247
  continue
248
 
249
  lower = sentence.lower()
250
 
251
- # reject repetitive phrases
252
  if any(bad in lower for bad in banned_phrases):
253
  continue
254
 
255
- # structure repetition detection
256
  structure = normalize_structure(sentence)
257
 
258
  if structure in used_structures:
@@ -260,22 +271,55 @@ def generate_ai_sentence(idiom, examples_map, used_structures):
260
 
261
  used_structures.add(structure)
262
 
 
 
 
 
 
 
 
 
263
  return sentence
264
 
265
  except Exception:
266
  continue
267
 
268
- # ---------- fallback ----------
269
  examples = examples_map.get(idiom.lower(), [])
270
 
271
- if examples:
272
- sentence = random.choice(examples)["en"]
273
 
274
- structure = normalize_structure(sentence)
 
275
 
276
- if structure not in used_structures:
277
- used_structures.add(structure)
278
- return sentence.replace(idiom, "_____")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
279
 
280
  return None
281
 
 
200
 
201
  return " ".join(words[:3])
202
 
 
203
  def generate_ai_sentence(idiom, examples_map, used_structures):
204
 
205
  subjects = [
 
222
 
223
  for _ in range(8):
224
 
225
+ try:
226
+
227
+ subject = random.choice(subjects)
228
+ tone = random.choice(tones)
229
 
230
+ prompt = f"""
231
+ Write one natural {tone} English sentence using the idiom "{idiom}".
232
+ Use "{subject}" as the subject.
233
+ Make it conversational and realistic.
234
+ """
235
 
 
236
  result = generator(
237
  prompt,
238
+ max_new_tokens=40,
239
  do_sample=True,
240
+ temperature=1.0,
241
+ top_k=50,
242
+ top_p=0.95,
243
+ repetition_penalty=1.2,
244
+ truncation=True
245
  )
 
 
246
 
247
+ sentence = result[0]["generated_text"]
248
+ sentence = sentence.replace(prompt, "").strip()
249
+
250
+ # Basic validation
251
  if not sentence:
252
  continue
253
 
254
+ if len(sentence.split()) < 5:
255
+ continue
256
+
257
  if idiom.lower() not in sentence.lower():
258
  continue
259
 
260
  lower = sentence.lower()
261
 
262
+ # Reject repetitive templates
263
  if any(bad in lower for bad in banned_phrases):
264
  continue
265
 
266
+ # Detect repeated structures
267
  structure = normalize_structure(sentence)
268
 
269
  if structure in used_structures:
 
271
 
272
  used_structures.add(structure)
273
 
274
+ # Replace idiom with blank
275
+ sentence = re.sub(
276
+ re.escape(idiom),
277
+ "_____",
278
+ sentence,
279
+ flags=re.IGNORECASE
280
+ )
281
+
282
  return sentence
283
 
284
  except Exception:
285
  continue
286
 
287
+ # ---------- FALLBACK TO DATASET ----------
288
  examples = examples_map.get(idiom.lower(), [])
289
 
290
+ valid_examples = []
 
291
 
292
+ for ex in examples:
293
+ text = ex.get("en", "")
294
 
295
+ if idiom.lower() in text.lower():
296
+
297
+ lower = text.lower()
298
+
299
+ if any(bad in lower for bad in banned_phrases):
300
+ continue
301
+
302
+ structure = normalize_structure(text)
303
+
304
+ if structure not in used_structures:
305
+ valid_examples.append(text)
306
+
307
+ if valid_examples:
308
+
309
+ sentence = random.choice(valid_examples)
310
+
311
+ used_structures.add(
312
+ normalize_structure(sentence)
313
+ )
314
+
315
+ sentence = re.sub(
316
+ re.escape(idiom),
317
+ "_____",
318
+ sentence,
319
+ flags=re.IGNORECASE
320
+ )
321
+
322
+ return sentence
323
 
324
  return None
325