# train.py - Training script for Hugging Face import os import torch import pandas as pd import numpy as np from datasets import load_dataset, DatasetDict from transformers import ( AutoTokenizer, AutoModelForSequenceClassification, TrainingArguments, Trainer, DataCollatorWithPadding ) import evaluate from sklearn.model_selection import train_test_split import json # Configuration CONFIG = { "model_name": "distilbert-base-uncased", "dataset_name": "6StringNinja/synthetic-servicenow-incidents", "output_dir": "./incident-classifier", "test_size": 0.2, "random_state": 42, "max_length": 128, "batch_size": 16, "learning_rate": 2e-5, "num_epochs": 3, "push_to_hub": True, "hub_model_id": "Rajeshwartiwari/incident-classification-model" } def load_and_prepare_data(): """Load and split dataset""" print("📂 Loading dataset...") dataset = load_dataset(CONFIG["dataset_name"]) # Convert to pandas for splitting df = pd.DataFrame(dataset["train"]) # Train/test split train_df, test_df = train_test_split( df, test_size=CONFIG["test_size"], random_state=CONFIG["random_state"], stratify=df["category"] ) # Convert back to Hugging Face datasets train_dataset = DatasetDict({"train": dataset["train"].new_from_pandas(train_df)}) test_dataset = DatasetDict({"test": dataset["train"].new_from_pandas(test_df)}) # Get labels labels = sorted(list(set(df["category"]))) label2id = {label: i for i, label in enumerate(labels)} id2label = {i: label for i, label in enumerate(labels)} print(f"✅ Dataset loaded. Categories: {labels}") print(f" Train samples: {len(train_df)}, Test samples: {len(test_df)}") return train_dataset["train"], test_dataset["test"], label2id, id2label def tokenize_function(examples, tokenizer): """Tokenize the examples""" # Combine short description and description texts = [ f"{sd} {d}" if d else sd for sd, d in zip(examples['short_description'], examples['description']) ] # Tokenize tokenized = tokenizer( texts, truncation=True, padding=True, max_length=CONFIG["max_length"] ) # Add labels tokenized["labels"] = [label2id[l] for l in examples["category"]] return tokenized def compute_metrics(eval_pred): """Compute evaluation metrics""" metric = evaluate.load("accuracy") logits, labels = eval_pred predictions = np.argmax(logits, axis=-1) # Calculate accuracy accuracy = metric.compute(predictions=predictions, references=labels) # You can add more metrics here return accuracy def main(): """Main training function""" print("🚀 Starting Incident Classification Model Training") # Load data train_dataset, test_dataset, label2id, id2label = load_and_prepare_data() # Initialize tokenizer and model print("🔧 Initializing tokenizer and model...") tokenizer = AutoTokenizer.from_pretrained(CONFIG["model_name"]) model = AutoModelForSequenceClassification.from_pretrained( CONFIG["model_name"], num_labels=len(label2id), id2label=id2label, label2id=label2id ) # Tokenize datasets print("🔠 Tokenizing datasets...") tokenized_train = train_dataset.map( lambda x: tokenize_function(x, tokenizer), batched=True, remove_columns=train_dataset.column_names ) tokenized_test = test_dataset.map( lambda x: tokenize_function(x, tokenizer), batched=True, remove_columns=test_dataset.column_names ) # Data collator data_collator = DataCollatorWithPadding(tokenizer=tokenizer) # Training arguments training_args = TrainingArguments( output_dir=CONFIG["output_dir"], learning_rate=CONFIG["learning_rate"], per_device_train_batch_size=CONFIG["batch_size"], per_device_eval_batch_size=CONFIG["batch_size"], num_train_epochs=CONFIG["num_epochs"], weight_decay=0.01, evaluation_strategy="epoch", save_strategy="epoch", load_best_model_at_end=True, metric_for_best_model="accuracy", report_to="none", # Disable WandB by default push_to_hub=CONFIG["push_to_hub"], hub_model_id=CONFIG["hub_model_id"], hub_strategy="end", save_total_limit=2, ) # Initialize trainer trainer = Trainer( model=model, args=training_args, train_dataset=tokenized_train, eval_dataset=tokenized_test, tokenizer=tokenizer, data_collator=data_collator, compute_metrics=compute_metrics, ) # Train print("🎯 Starting training...") trainer.train() # Evaluate print("📊 Evaluating model...") eval_results = trainer.evaluate() print(f"✅ Evaluation results: {eval_results}") # Save everything locally print("💾 Saving model locally...") trainer.save_model(CONFIG["output_dir"]) tokenizer.save_pretrained(CONFIG["output_dir"]) # Save label mappings with open(os.path.join(CONFIG["output_dir"], "label_mappings.json"), "w") as f: json.dump({"label2id": label2id, "id2label": id2label}, f, indent=2) # Save configuration with open(os.path.join(CONFIG["output_dir"], "config.json"), "w") as f: json.dump(CONFIG, f, indent=2) print(f"🎉 Training complete! Model saved to {CONFIG['output_dir']}") if CONFIG["push_to_hub"]: print("☁️ Pushing to Hugging Face Hub...") trainer.push_to_hub() tokenizer.push_to_hub(CONFIG["hub_model_id"]) print(f"✅ Model pushed to: https://huggingface.co/{CONFIG['hub_model_id']}") if __name__ == "__main__": main()