jefffffff9 Claude Sonnet 4.6 commited on
Commit
bfa9d46
·
1 Parent(s): 39604b3

Fix Cell 8: correct WaxalNLP subset names + drop fleurs fallback

Browse files

Two bugs:
1. WaxalNLP has no 'bam' subset — Bambara is absent entirely.
Correct Fula subset is 'ful_asr' (not 'ful').
WAXAL_SUBSET_MAP: bam→None (skip), ful→'ful_asr'.

2. google/fleurs uses a dataset script (fleurs.py) which datasets>=3.0
refuses to execute. Removed fleurs fallback entirely.

Cell 4 updated: Common Voice Bambara ('bm') and Fula ('ff') enabled
by default — Common Voice is now the primary large-scale source for
Bambara since WaxalNLP has no bam subset.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>

notebooks/kaggle_master_trainer.ipynb CHANGED
@@ -49,7 +49,7 @@
49
  "id": "cell-ext-config",
50
  "metadata": {},
51
  "outputs": [],
52
- "source": "# ── Cell 4: External dataset configuration ───────────────────────────────────\n# Add or remove entries here to include additional HF datasets.\n# Each entry: (repo_id, config_name, split, text_column, language_filter_or_None)\n#\n# Set ENABLED=True to activate a source, False to skip it.\n\nEXTERNAL_DATASETS = [\n {\n 'enabled' : False, # ← set True to include Common Voice\n 'repo_id' : 'mozilla-foundation/common_voice_13_0',\n 'config' : 'bm', # Bambara config code in Common Voice\n 'split' : 'train',\n 'text_col' : 'sentence',\n 'lang' : 'bam',\n 'max_samples': 2_000,\n },\n {\n 'enabled' : False,\n 'repo_id' : 'mozilla-foundation/common_voice_13_0',\n 'config' : 'ff', # Fula/Fulah config code\n 'split' : 'train',\n 'text_col' : 'sentence',\n 'lang' : 'ful',\n 'max_samples': 2_000,\n },\n # Add more datasets here:\n # {\n # 'enabled': True,\n # 'repo_id': 'your/dataset',\n # 'config': 'subset_name',\n # 'split': 'train',\n # 'text_col':'transcription',\n # 'lang': 'bam',\n # 'max_samples': 1000,\n # },\n]\n\nprint(f'External sources configured: {len(EXTERNAL_DATASETS)}')\nprint(f'Active for lang={TRAIN_LANG}: {sum(1 for d in EXTERNAL_DATASETS if d[\"enabled\"] and d[\"lang\"] == TRAIN_LANG)}')"
53
  },
54
  {
55
  "cell_type": "code",
@@ -81,7 +81,7 @@
81
  "id": "cell-waxal",
82
  "metadata": {},
83
  "outputs": [],
84
- "source": "# ── Cell 8: Load WaxalNLP (with FLEURS fallback) ──────────────────────────────\n# WaxalNLP subsets for ASR: 'bam' (Bambara), 'ful' (Fula)\n# If the ASR subsets are missing, falls back to google/fleurs which has the\n# same language data under different subset codes.\n\nfrom datasets import load_dataset, Audio as HFAudio\n\nWAXAL_SUBSET_MAP = {'bam': 'bam', 'ful': 'ful'}\nFLEURS_SUBSET_MAP = {'bam': 'bam_ML', 'ful': 'ff_SN'}\n\nwaxal_ds = None\n\n# Try WaxalNLP first\ntry:\n subset = WAXAL_SUBSET_MAP[TRAIN_LANG]\n print(f'Loading google/WaxalNLP subset={subset} (streaming) ...')\n waxal_ds = load_dataset(\n 'google/WaxalNLP', subset,\n split='train', streaming=True,\n token=HF_TOKEN,\n )\n # Probe one item to verify it has audio + text\n probe = next(iter(waxal_ds))\n text_key = next(\n (k for k in ['transcription', 'text', 'sentence', 'normalized_text'] if k in probe),\n None\n )\n if text_key is None or 'audio' not in probe:\n raise ValueError(f'WaxalNLP/{subset} has unexpected schema: {list(probe.keys())}')\n print(f'WaxalNLP/{subset} ready — text column: \"{text_key}\"')\n WAXAL_TEXT_COL = text_key\nexcept Exception as e:\n print(f'WaxalNLP not available ({e}) — falling back to google/fleurs')\n subset = FLEURS_SUBSET_MAP[TRAIN_LANG]\n waxal_ds = load_dataset(\n 'google/fleurs', subset,\n split='train', streaming=True,\n token=HF_TOKEN,\n )\n probe = next(iter(waxal_ds))\n WAXAL_TEXT_COL = next(\n (k for k in ['transcription', 'text', 'sentence', 'raw_transcription'] if k in probe),\n list(probe.keys())[0]\n )\n print(f'FLEURS/{subset} ready — text column: \"{WAXAL_TEXT_COL}\"')\n\nprint(f'\\nWaxal/FLEURS source ready for {TRAIN_LANG}')"
85
  },
86
  {
87
  "cell_type": "code",
 
49
  "id": "cell-ext-config",
50
  "metadata": {},
51
  "outputs": [],
52
+ "source": "# ── Cell 4: External dataset configuration ───────────────────────────────────\n# Add or remove entries to include additional HF datasets.\n# Each entry: repo_id, config (language code), split, text column, lang tag, max samples.\n#\n# Common Voice Bambara ('bm') is enabled by default — it is the primary\n# large-scale Bambara ASR source since WaxalNLP has no Bambara subset.\n# Common Voice Fula ('ff') is enabled as a supplement to WaxalNLP ful_asr.\n#\n# ⚠️ Common Voice requires accepting the dataset terms on HuggingFace:\n# https://huggingface.co/datasets/mozilla-foundation/common_voice_17_0\n# Your HF_TOKEN must belong to an account that has accepted them.\n\nEXTERNAL_DATASETS = [\n {\n 'enabled' : True, # PRIMARY Bambara source (WaxalNLP has no bam)\n 'repo_id' : 'mozilla-foundation/common_voice_17_0',\n 'config' : 'bm', # Bambara\n 'split' : 'train',\n 'text_col' : 'sentence',\n 'lang' : 'bam',\n 'max_samples': 5_000,\n },\n {\n 'enabled' : True, # Supplement to WaxalNLP ful_asr\n 'repo_id' : 'mozilla-foundation/common_voice_17_0',\n 'config' : 'ff', # Fulah / Fula\n 'split' : 'train',\n 'text_col' : 'sentence',\n 'lang' : 'ful',\n 'max_samples': 3_000,\n },\n # Add more datasets here — any HF dataset with an 'audio' column works:\n # {\n # 'enabled' : False,\n # 'repo_id' : 'your/dataset',\n # 'config' : 'subset_name',\n # 'split' : 'train',\n # 'text_col' : 'transcription',\n # 'lang' : 'bam',\n # 'max_samples': 2_000,\n # },\n]\n\nactive = [d for d in EXTERNAL_DATASETS if d['enabled'] and d['lang'] == TRAIN_LANG]\nprint(f'External sources configured : {len(EXTERNAL_DATASETS)}')\nprint(f'Active for lang={TRAIN_LANG} : {len(active)}')\nfor d in active:\n print(f' • {d[\"repo_id\"]} / {d[\"config\"]} — up to {d[\"max_samples\"]} samples')"
53
  },
54
  {
55
  "cell_type": "code",
 
81
  "id": "cell-waxal",
82
  "metadata": {},
83
  "outputs": [],
84
+ "source": "# ── Cell 8: Load WaxalNLP ─────────────────────────────────────────────────────\n# WaxalNLP (google/WaxalNLP) available ASR subsets:\n# Fula → 'ful_asr' ✅ (confirmed in dataset card)\n# Bambara → NOT present ❌ (bam is not a WaxalNLP config)\n#\n# For Bambara, user corrections + Common Voice (Cell 4) are the data sources.\n# google/fleurs is NOT used — datasets>=3.0 refuses to run its script loader.\n#\n# Reference: WaxalNLP configs include ful_asr, ful_tts, ewe_asr, hau_tts, etc.\n# Full list: https://huggingface.co/datasets/google/WaxalNLP\n\nfrom datasets import load_dataset, Audio as HFAudio\n\nWAXAL_SUBSET_MAP = {\n 'bam': None, # no Bambara subset in WaxalNLP\n 'ful': 'ful_asr', # confirmed available config\n}\n\nwaxal_ds = None\nWAXAL_TEXT_COL = 'transcription'\n\nsubset = WAXAL_SUBSET_MAP.get(TRAIN_LANG)\n\nif subset is None:\n print(f'WaxalNLP: no subset available for lang={TRAIN_LANG}.')\n print(' Bambara training will rely on user corrections + any enabled external datasets (Cell 4).')\nelse:\n try:\n print(f'Loading google/WaxalNLP subset={subset} (streaming) ...')\n waxal_ds = load_dataset(\n 'google/WaxalNLP', subset,\n split='train',\n streaming=True,\n token=HF_TOKEN,\n )\n\n # Verify schema by probing the first item\n probe = next(iter(waxal_ds))\n if 'audio' not in probe:\n raise ValueError(f'No audio column. Available: {list(probe.keys())}')\n\n WAXAL_TEXT_COL = next(\n (k for k in ['transcription', 'text', 'sentence', 'normalized_text']\n if k in probe),\n None,\n )\n if WAXAL_TEXT_COL is None:\n raise ValueError(f'No text column found. Keys: {list(probe.keys())}')\n\n print(f'✅ WaxalNLP/{subset} ready — text column: \"{WAXAL_TEXT_COL}\"')\n\n except Exception as e:\n print(f'⚠️ WaxalNLP/{subset} failed: {e}')\n print(' Continuing without WaxalNLP — enable Common Voice in Cell 4 for coverage.')\n waxal_ds = None\n\nprint()\nstatus = f'WaxalNLP/{subset}' if waxal_ds is not None else 'not available'\nprint(f'Waxal source for {TRAIN_LANG}: {status}')"
85
  },
86
  {
87
  "cell_type": "code",