Spaces:
Sleeping
Sleeping
Download token_train.py from ArushBuilds/Pragya: direct link, hf CLI and curl.
- Browser
- Download file 446 Bytes
-
https://huggingface.co/spaces/ArushBuilds/Pragya/resolve/2f5d2dceb9fe8d9b6a898c273e6c62508ba3b233/token_train.py
- Command line
-
hf download hf://spaces/ArushBuilds/Pragya@2f5d2dceb9fe8d9b6a898c273e6c62508ba3b233/token_train.py
-
curl -L -o token_train.py https://huggingface.co/spaces/ArushBuilds/Pragya/resolve/2f5d2dceb9fe8d9b6a898c273e6c62508ba3b233/token_train.py
446 Bytes
| import os | |
| from tokenizer import train_sentencepiece | |
| from config import vocab_size | |
| DATA_DIR = "./training_data" | |
| text_files = [ | |
| os.path.join(DATA_DIR, f) | |
| for f in os.listdir(DATA_DIR) | |
| if f.endswith(".txt") | |
| ] | |
| print(f"Found {len(text_files)} text files") | |
| if not text_files: | |
| raise ValueError( | |
| f"No .txt files found in {DATA_DIR}" | |
| ) | |
| tokenizer = train_sentencepiece(text_files,vocab_size=vocab_size) | |