Shilpi Kumari
Add application file
a1ca7f2
Raw History Blame Contribute Delete
5.92 kB
# train.py - Training script for Hugging Face
import os
import torch
import pandas as pd
import numpy as np
from datasets import load_dataset, DatasetDict
from transformers import (
AutoTokenizer,
AutoModelForSequenceClassification,
TrainingArguments,
Trainer,
DataCollatorWithPadding
)
import evaluate
from sklearn.model_selection import train_test_split
import json
# Configuration
CONFIG = {
"model_name": "distilbert-base-uncased",
"dataset_name": "6StringNinja/synthetic-servicenow-incidents",
"output_dir": "./incident-classifier",
"test_size": 0.2,
"random_state": 42,
"max_length": 128,
"batch_size": 16,
"learning_rate": 2e-5,
"num_epochs": 3,
"push_to_hub": True,
"hub_model_id": "Rajeshwartiwari/incident-classification-model"
}
def load_and_prepare_data():
"""Load and split dataset"""
print("📂 Loading dataset...")
dataset = load_dataset(CONFIG["dataset_name"])
# Convert to pandas for splitting
df = pd.DataFrame(dataset["train"])
# Train/test split
train_df, test_df = train_test_split(
df,
test_size=CONFIG["test_size"],
random_state=CONFIG["random_state"],
stratify=df["category"]
)
# Convert back to Hugging Face datasets
train_dataset = DatasetDict({"train": dataset["train"].new_from_pandas(train_df)})
test_dataset = DatasetDict({"test": dataset["train"].new_from_pandas(test_df)})
# Get labels
labels = sorted(list(set(df["category"])))
label2id = {label: i for i, label in enumerate(labels)}
id2label = {i: label for i, label in enumerate(labels)}
print(f"✅ Dataset loaded. Categories: {labels}")
print(f" Train samples: {len(train_df)}, Test samples: {len(test_df)}")
return train_dataset["train"], test_dataset["test"], label2id, id2label
def tokenize_function(examples, tokenizer):
"""Tokenize the examples"""
# Combine short description and description
texts = [
f"{sd} {d}" if d else sd
for sd, d in zip(examples['short_description'], examples['description'])
]
# Tokenize
tokenized = tokenizer(
texts,
truncation=True,
padding=True,
max_length=CONFIG["max_length"]
)
# Add labels
tokenized["labels"] = [label2id[l] for l in examples["category"]]
return tokenized
def compute_metrics(eval_pred):
"""Compute evaluation metrics"""
metric = evaluate.load("accuracy")
logits, labels = eval_pred
predictions = np.argmax(logits, axis=-1)
# Calculate accuracy
accuracy = metric.compute(predictions=predictions, references=labels)
# You can add more metrics here
return accuracy
def main():
"""Main training function"""
print("🚀 Starting Incident Classification Model Training")
# Load data
train_dataset, test_dataset, label2id, id2label = load_and_prepare_data()
# Initialize tokenizer and model
print("🔧 Initializing tokenizer and model...")
tokenizer = AutoTokenizer.from_pretrained(CONFIG["model_name"])
model = AutoModelForSequenceClassification.from_pretrained(
CONFIG["model_name"],
num_labels=len(label2id),
id2label=id2label,
label2id=label2id
)
# Tokenize datasets
print("🔠 Tokenizing datasets...")
tokenized_train = train_dataset.map(
lambda x: tokenize_function(x, tokenizer),
batched=True,
remove_columns=train_dataset.column_names
)
tokenized_test = test_dataset.map(
lambda x: tokenize_function(x, tokenizer),
batched=True,
remove_columns=test_dataset.column_names
)
# Data collator
data_collator = DataCollatorWithPadding(tokenizer=tokenizer)
# Training arguments
training_args = TrainingArguments(
output_dir=CONFIG["output_dir"],
learning_rate=CONFIG["learning_rate"],
per_device_train_batch_size=CONFIG["batch_size"],
per_device_eval_batch_size=CONFIG["batch_size"],
num_train_epochs=CONFIG["num_epochs"],
weight_decay=0.01,
evaluation_strategy="epoch",
save_strategy="epoch",
load_best_model_at_end=True,
metric_for_best_model="accuracy",
report_to="none", # Disable WandB by default
push_to_hub=CONFIG["push_to_hub"],
hub_model_id=CONFIG["hub_model_id"],
hub_strategy="end",
save_total_limit=2,
)
# Initialize trainer
trainer = Trainer(
model=model,
args=training_args,
train_dataset=tokenized_train,
eval_dataset=tokenized_test,
tokenizer=tokenizer,
data_collator=data_collator,
compute_metrics=compute_metrics,
)
# Train
print("🎯 Starting training...")
trainer.train()
# Evaluate
print("📊 Evaluating model...")
eval_results = trainer.evaluate()
print(f"✅ Evaluation results: {eval_results}")
# Save everything locally
print("💾 Saving model locally...")
trainer.save_model(CONFIG["output_dir"])
tokenizer.save_pretrained(CONFIG["output_dir"])
# Save label mappings
with open(os.path.join(CONFIG["output_dir"], "label_mappings.json"), "w") as f:
json.dump({"label2id": label2id, "id2label": id2label}, f, indent=2)
# Save configuration
with open(os.path.join(CONFIG["output_dir"], "config.json"), "w") as f:
json.dump(CONFIG, f, indent=2)
print(f"🎉 Training complete! Model saved to {CONFIG['output_dir']}")
if CONFIG["push_to_hub"]:
print("☁️ Pushing to Hugging Face Hub...")
trainer.push_to_hub()
tokenizer.push_to_hub(CONFIG["hub_model_id"])
print(f"✅ Model pushed to: https://huggingface.co/{CONFIG['hub_model_id']}")
if __name__ == "__main__":
main()