Download train_model.py from aaronmrls/ml-api: direct link, hf CLI and curl.
- Browser
- Download file 2.43 kB
-
https://huggingface.co/spaces/aaronmrls/ml-api/resolve/main/train_model.py
- Command line
-
hf download hf://spaces/aaronmrls/ml-api/train_model.py
-
curl -L -o train_model.py https://huggingface.co/spaces/aaronmrls/ml-api/resolve/main/train_model.py
2.43 kB
| import pandas as pd | |
| import torch | |
| from transformers import ( | |
| AutoTokenizer, | |
| AutoModelForSequenceClassification, | |
| Trainer, | |
| TrainingArguments, | |
| ) | |
| from sklearn.model_selection import train_test_split | |
| from sklearn.metrics import accuracy_score | |
| import numpy as np | |
| from datasets import Dataset, DatasetDict | |
| # Load dataset | |
| df = pd.read_csv("ml/data/request_data.csv") | |
| # Fill missing comments | |
| df["comments"] = df["comments"].fillna("") | |
| # Combine into single text field | |
| df["text"] = ( | |
| df["category"].fillna("") + " - " + | |
| df["subcategory"].fillna("") + " in " + | |
| df["area"].fillna("") + ". " + | |
| df["comments"].fillna("") | |
| ) | |
| # Ensure numeric priority labels (0β5) | |
| df["priority"] = pd.to_numeric(df["priority"], errors="coerce").fillna(0).astype(int) | |
| df = df[["text", "priority"]] | |
| # Split train/test | |
| train_df, test_df = train_test_split(df, test_size=0.2, random_state=42) | |
| dataset = DatasetDict({ | |
| "train": Dataset.from_pandas(train_df), | |
| "test": Dataset.from_pandas(test_df) | |
| }) | |
| # π USE MULTILINGUAL TOKENIZER | |
| model_name = "distilbert-base-multilingual-cased" | |
| tokenizer = AutoTokenizer.from_pretrained(model_name) | |
| def tokenize(batch): | |
| tokenized = tokenizer(batch["text"], padding=True, truncation=True) | |
| tokenized["labels"] = batch["priority"] | |
| return tokenized | |
| dataset = dataset.map(tokenize, batched=True) | |
| # π USE MULTILINGUAL MODEL | |
| model = AutoModelForSequenceClassification.from_pretrained(model_name, num_labels=6) | |
| # Metric | |
| def compute_metrics(eval_pred): | |
| logits, labels = eval_pred | |
| preds = np.argmax(logits, axis=1) | |
| return {"accuracy": accuracy_score(labels, preds)} | |
| # Training arguments | |
| training_args = TrainingArguments( | |
| output_dir="./ml/model", | |
| evaluation_strategy="epoch", | |
| logging_dir="./ml/model/logs", | |
| logging_steps=10, | |
| load_best_model_at_end=True, | |
| metric_for_best_model="accuracy", | |
| greater_is_better=True, | |
| per_device_train_batch_size=8, | |
| per_device_eval_batch_size=8, | |
| num_train_epochs=3, | |
| save_total_limit=2, | |
| ) | |
| # Trainer | |
| trainer = Trainer( | |
| model=model, | |
| args=training_args, | |
| train_dataset=dataset["train"], | |
| eval_dataset=dataset["test"], | |
| tokenizer=tokenizer, | |
| compute_metrics=compute_metrics | |
| ) | |
| # Train | |
| trainer.train() | |
| # Save model + tokenizer | |
| model.save_pretrained("./ml/model") | |
| tokenizer.save_pretrained("./ml/model/tokenizer") | |
| print("β Multilingual model trained and saved to ./ml/model/") | |