Download train.py from Rajeshwartiwari/incident-classification-system: direct link, hf CLI and curl.
- Browser
- Download file 5.92 kB
-
https://huggingface.co/spaces/Rajeshwartiwari/incident-classification-system/resolve/main/train.py
- Command line
-
hf download hf://spaces/Rajeshwartiwari/incident-classification-system/train.py
-
curl -L -o train.py https://huggingface.co/spaces/Rajeshwartiwari/incident-classification-system/resolve/main/train.py
5.92 kB
| # train.py - Training script for Hugging Face | |
| import os | |
| import torch | |
| import pandas as pd | |
| import numpy as np | |
| from datasets import load_dataset, DatasetDict | |
| from transformers import ( | |
| AutoTokenizer, | |
| AutoModelForSequenceClassification, | |
| TrainingArguments, | |
| Trainer, | |
| DataCollatorWithPadding | |
| ) | |
| import evaluate | |
| from sklearn.model_selection import train_test_split | |
| import json | |
| # Configuration | |
| CONFIG = { | |
| "model_name": "distilbert-base-uncased", | |
| "dataset_name": "6StringNinja/synthetic-servicenow-incidents", | |
| "output_dir": "./incident-classifier", | |
| "test_size": 0.2, | |
| "random_state": 42, | |
| "max_length": 128, | |
| "batch_size": 16, | |
| "learning_rate": 2e-5, | |
| "num_epochs": 3, | |
| "push_to_hub": True, | |
| "hub_model_id": "Rajeshwartiwari/incident-classification-model" | |
| } | |
| def load_and_prepare_data(): | |
| """Load and split dataset""" | |
| print("📂 Loading dataset...") | |
| dataset = load_dataset(CONFIG["dataset_name"]) | |
| # Convert to pandas for splitting | |
| df = pd.DataFrame(dataset["train"]) | |
| # Train/test split | |
| train_df, test_df = train_test_split( | |
| df, | |
| test_size=CONFIG["test_size"], | |
| random_state=CONFIG["random_state"], | |
| stratify=df["category"] | |
| ) | |
| # Convert back to Hugging Face datasets | |
| train_dataset = DatasetDict({"train": dataset["train"].new_from_pandas(train_df)}) | |
| test_dataset = DatasetDict({"test": dataset["train"].new_from_pandas(test_df)}) | |
| # Get labels | |
| labels = sorted(list(set(df["category"]))) | |
| label2id = {label: i for i, label in enumerate(labels)} | |
| id2label = {i: label for i, label in enumerate(labels)} | |
| print(f"✅ Dataset loaded. Categories: {labels}") | |
| print(f" Train samples: {len(train_df)}, Test samples: {len(test_df)}") | |
| return train_dataset["train"], test_dataset["test"], label2id, id2label | |
| def tokenize_function(examples, tokenizer): | |
| """Tokenize the examples""" | |
| # Combine short description and description | |
| texts = [ | |
| f"{sd} {d}" if d else sd | |
| for sd, d in zip(examples['short_description'], examples['description']) | |
| ] | |
| # Tokenize | |
| tokenized = tokenizer( | |
| texts, | |
| truncation=True, | |
| padding=True, | |
| max_length=CONFIG["max_length"] | |
| ) | |
| # Add labels | |
| tokenized["labels"] = [label2id[l] for l in examples["category"]] | |
| return tokenized | |
| def compute_metrics(eval_pred): | |
| """Compute evaluation metrics""" | |
| metric = evaluate.load("accuracy") | |
| logits, labels = eval_pred | |
| predictions = np.argmax(logits, axis=-1) | |
| # Calculate accuracy | |
| accuracy = metric.compute(predictions=predictions, references=labels) | |
| # You can add more metrics here | |
| return accuracy | |
| def main(): | |
| """Main training function""" | |
| print("🚀 Starting Incident Classification Model Training") | |
| # Load data | |
| train_dataset, test_dataset, label2id, id2label = load_and_prepare_data() | |
| # Initialize tokenizer and model | |
| print("🔧 Initializing tokenizer and model...") | |
| tokenizer = AutoTokenizer.from_pretrained(CONFIG["model_name"]) | |
| model = AutoModelForSequenceClassification.from_pretrained( | |
| CONFIG["model_name"], | |
| num_labels=len(label2id), | |
| id2label=id2label, | |
| label2id=label2id | |
| ) | |
| # Tokenize datasets | |
| print("🔠 Tokenizing datasets...") | |
| tokenized_train = train_dataset.map( | |
| lambda x: tokenize_function(x, tokenizer), | |
| batched=True, | |
| remove_columns=train_dataset.column_names | |
| ) | |
| tokenized_test = test_dataset.map( | |
| lambda x: tokenize_function(x, tokenizer), | |
| batched=True, | |
| remove_columns=test_dataset.column_names | |
| ) | |
| # Data collator | |
| data_collator = DataCollatorWithPadding(tokenizer=tokenizer) | |
| # Training arguments | |
| training_args = TrainingArguments( | |
| output_dir=CONFIG["output_dir"], | |
| learning_rate=CONFIG["learning_rate"], | |
| per_device_train_batch_size=CONFIG["batch_size"], | |
| per_device_eval_batch_size=CONFIG["batch_size"], | |
| num_train_epochs=CONFIG["num_epochs"], | |
| weight_decay=0.01, | |
| evaluation_strategy="epoch", | |
| save_strategy="epoch", | |
| load_best_model_at_end=True, | |
| metric_for_best_model="accuracy", | |
| report_to="none", # Disable WandB by default | |
| push_to_hub=CONFIG["push_to_hub"], | |
| hub_model_id=CONFIG["hub_model_id"], | |
| hub_strategy="end", | |
| save_total_limit=2, | |
| ) | |
| # Initialize trainer | |
| trainer = Trainer( | |
| model=model, | |
| args=training_args, | |
| train_dataset=tokenized_train, | |
| eval_dataset=tokenized_test, | |
| tokenizer=tokenizer, | |
| data_collator=data_collator, | |
| compute_metrics=compute_metrics, | |
| ) | |
| # Train | |
| print("🎯 Starting training...") | |
| trainer.train() | |
| # Evaluate | |
| print("📊 Evaluating model...") | |
| eval_results = trainer.evaluate() | |
| print(f"✅ Evaluation results: {eval_results}") | |
| # Save everything locally | |
| print("💾 Saving model locally...") | |
| trainer.save_model(CONFIG["output_dir"]) | |
| tokenizer.save_pretrained(CONFIG["output_dir"]) | |
| # Save label mappings | |
| with open(os.path.join(CONFIG["output_dir"], "label_mappings.json"), "w") as f: | |
| json.dump({"label2id": label2id, "id2label": id2label}, f, indent=2) | |
| # Save configuration | |
| with open(os.path.join(CONFIG["output_dir"], "config.json"), "w") as f: | |
| json.dump(CONFIG, f, indent=2) | |
| print(f"🎉 Training complete! Model saved to {CONFIG['output_dir']}") | |
| if CONFIG["push_to_hub"]: | |
| print("☁️ Pushing to Hugging Face Hub...") | |
| trainer.push_to_hub() | |
| tokenizer.push_to_hub(CONFIG["hub_model_id"]) | |
| print(f"✅ Model pushed to: https://huggingface.co/{CONFIG['hub_model_id']}") | |
| if __name__ == "__main__": | |
| main() |