Spaces:
Sleeping
Sleeping
jefffffff9 Claude Sonnet 4.6 commited on
Commit ·
bfa9d46
1
Parent(s): 39604b3
Fix Cell 8: correct WaxalNLP subset names + drop fleurs fallback
Browse filesTwo bugs:
1. WaxalNLP has no 'bam' subset — Bambara is absent entirely.
Correct Fula subset is 'ful_asr' (not 'ful').
WAXAL_SUBSET_MAP: bam→None (skip), ful→'ful_asr'.
2. google/fleurs uses a dataset script (fleurs.py) which datasets>=3.0
refuses to execute. Removed fleurs fallback entirely.
Cell 4 updated: Common Voice Bambara ('bm') and Fula ('ff') enabled
by default — Common Voice is now the primary large-scale source for
Bambara since WaxalNLP has no bam subset.
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
notebooks/kaggle_master_trainer.ipynb
CHANGED
|
@@ -49,7 +49,7 @@
|
|
| 49 |
"id": "cell-ext-config",
|
| 50 |
"metadata": {},
|
| 51 |
"outputs": [],
|
| 52 |
-
"source": "# ── Cell 4: External dataset configuration ───────────────────────────────────\n# Add or remove entries
|
| 53 |
},
|
| 54 |
{
|
| 55 |
"cell_type": "code",
|
|
@@ -81,7 +81,7 @@
|
|
| 81 |
"id": "cell-waxal",
|
| 82 |
"metadata": {},
|
| 83 |
"outputs": [],
|
| 84 |
-
"source": "# ── Cell 8: Load WaxalNLP
|
| 85 |
},
|
| 86 |
{
|
| 87 |
"cell_type": "code",
|
|
|
|
| 49 |
"id": "cell-ext-config",
|
| 50 |
"metadata": {},
|
| 51 |
"outputs": [],
|
| 52 |
+
"source": "# ── Cell 4: External dataset configuration ───────────────────────────────────\n# Add or remove entries to include additional HF datasets.\n# Each entry: repo_id, config (language code), split, text column, lang tag, max samples.\n#\n# Common Voice Bambara ('bm') is enabled by default — it is the primary\n# large-scale Bambara ASR source since WaxalNLP has no Bambara subset.\n# Common Voice Fula ('ff') is enabled as a supplement to WaxalNLP ful_asr.\n#\n# ⚠️ Common Voice requires accepting the dataset terms on HuggingFace:\n# https://huggingface.co/datasets/mozilla-foundation/common_voice_17_0\n# Your HF_TOKEN must belong to an account that has accepted them.\n\nEXTERNAL_DATASETS = [\n {\n 'enabled' : True, # PRIMARY Bambara source (WaxalNLP has no bam)\n 'repo_id' : 'mozilla-foundation/common_voice_17_0',\n 'config' : 'bm', # Bambara\n 'split' : 'train',\n 'text_col' : 'sentence',\n 'lang' : 'bam',\n 'max_samples': 5_000,\n },\n {\n 'enabled' : True, # Supplement to WaxalNLP ful_asr\n 'repo_id' : 'mozilla-foundation/common_voice_17_0',\n 'config' : 'ff', # Fulah / Fula\n 'split' : 'train',\n 'text_col' : 'sentence',\n 'lang' : 'ful',\n 'max_samples': 3_000,\n },\n # Add more datasets here — any HF dataset with an 'audio' column works:\n # {\n # 'enabled' : False,\n # 'repo_id' : 'your/dataset',\n # 'config' : 'subset_name',\n # 'split' : 'train',\n # 'text_col' : 'transcription',\n # 'lang' : 'bam',\n # 'max_samples': 2_000,\n # },\n]\n\nactive = [d for d in EXTERNAL_DATASETS if d['enabled'] and d['lang'] == TRAIN_LANG]\nprint(f'External sources configured : {len(EXTERNAL_DATASETS)}')\nprint(f'Active for lang={TRAIN_LANG} : {len(active)}')\nfor d in active:\n print(f' • {d[\"repo_id\"]} / {d[\"config\"]} — up to {d[\"max_samples\"]} samples')"
|
| 53 |
},
|
| 54 |
{
|
| 55 |
"cell_type": "code",
|
|
|
|
| 81 |
"id": "cell-waxal",
|
| 82 |
"metadata": {},
|
| 83 |
"outputs": [],
|
| 84 |
+
"source": "# ── Cell 8: Load WaxalNLP ─────────────────────────────────────────────────────\n# WaxalNLP (google/WaxalNLP) available ASR subsets:\n# Fula → 'ful_asr' ✅ (confirmed in dataset card)\n# Bambara → NOT present ❌ (bam is not a WaxalNLP config)\n#\n# For Bambara, user corrections + Common Voice (Cell 4) are the data sources.\n# google/fleurs is NOT used — datasets>=3.0 refuses to run its script loader.\n#\n# Reference: WaxalNLP configs include ful_asr, ful_tts, ewe_asr, hau_tts, etc.\n# Full list: https://huggingface.co/datasets/google/WaxalNLP\n\nfrom datasets import load_dataset, Audio as HFAudio\n\nWAXAL_SUBSET_MAP = {\n 'bam': None, # no Bambara subset in WaxalNLP\n 'ful': 'ful_asr', # confirmed available config\n}\n\nwaxal_ds = None\nWAXAL_TEXT_COL = 'transcription'\n\nsubset = WAXAL_SUBSET_MAP.get(TRAIN_LANG)\n\nif subset is None:\n print(f'WaxalNLP: no subset available for lang={TRAIN_LANG}.')\n print(' Bambara training will rely on user corrections + any enabled external datasets (Cell 4).')\nelse:\n try:\n print(f'Loading google/WaxalNLP subset={subset} (streaming) ...')\n waxal_ds = load_dataset(\n 'google/WaxalNLP', subset,\n split='train',\n streaming=True,\n token=HF_TOKEN,\n )\n\n # Verify schema by probing the first item\n probe = next(iter(waxal_ds))\n if 'audio' not in probe:\n raise ValueError(f'No audio column. Available: {list(probe.keys())}')\n\n WAXAL_TEXT_COL = next(\n (k for k in ['transcription', 'text', 'sentence', 'normalized_text']\n if k in probe),\n None,\n )\n if WAXAL_TEXT_COL is None:\n raise ValueError(f'No text column found. Keys: {list(probe.keys())}')\n\n print(f'✅ WaxalNLP/{subset} ready — text column: \"{WAXAL_TEXT_COL}\"')\n\n except Exception as e:\n print(f'⚠️ WaxalNLP/{subset} failed: {e}')\n print(' Continuing without WaxalNLP — enable Common Voice in Cell 4 for coverage.')\n waxal_ds = None\n\nprint()\nstatus = f'WaxalNLP/{subset}' if waxal_ds is not None else 'not available'\nprint(f'Waxal source for {TRAIN_LANG}: {status}')"
|
| 85 |
},
|
| 86 |
{
|
| 87 |
"cell_type": "code",
|