CE_5.0 / source_code.py
mherrador's picture
Create source_code.py
6d70cf6 verified
Raw
History Blame Contribute Delete
2.87 kB
# Install necessary dependencies
!pip install transformers datasets torch accelerate huggingface_hub sentencepiece
!pip install ctranslate2
!pip install -U huggingface_hub
import json
import torch
import os
from transformers import AutoTokenizer, AutoModelForSeq2SeqLM, Trainer, TrainingArguments
from datasets import Dataset
from huggingface_hub import HfApi, create_repo, upload_file
# Load and prepare the dataset
with open('CE5.0_Expert.json', 'r') as f:
data = json.load(f)
dataset = Dataset.from_dict({
'input': [item['input'] for item in data],
'output': [item['output'] for item in data]
})
# Load tokenizer and model
model_name = "google/flan-t5-small"
tokenizer = AutoTokenizer.from_pretrained(model_name)
model = AutoModelForSeq2SeqLM.from_pretrained(model_name)
# Tokenize the dataset
def preprocess_function(examples):
inputs = tokenizer(examples["input"], max_length=512, truncation=True, padding="max_length")
outputs = tokenizer(examples["output"], max_length=512, truncation=True, padding="max_length")
return {
"input_ids": inputs.input_ids,
"attention_mask": inputs.attention_mask,
"labels": outputs.input_ids,
}
tokenized_dataset = dataset.map(preprocess_function, batched=True)
# Define training arguments
training_args = TrainingArguments(
output_dir="./results",
num_train_epochs=3,
per_device_train_batch_size=8,
save_steps=100,
save_total_limit=2,
)
# Initialize Trainer
trainer = Trainer(
model=model,
args=training_args,
train_dataset=tokenized_dataset,
)
# Train the model
trainer.train()
# Save the model and tokenizer
model.save_pretrained("./checkpoint")
tokenizer.save_pretrained("./checkpoint")
# Convert to CTranslate2 format
import ctranslate2
converter = ctranslate2.converters.TransformersConverter("./checkpoint")
converter.convert("./ctranslate2_model", quantization="int8")
# Push to Hugging Face
hf_token = "HF_TOKEN" # Make sure to keep this secret
model_name = "CE_5.0"
api = HfApi()
# Create the repository
create_repo(model_name, private=True, token=hf_token)
# Upload the CTranslate2 model files
for root, dirs, files in os.walk("./ctranslate2_model"):
for file in files:
file_path = os.path.join(root, file)
api.upload_file(
path_or_fileobj=file_path,
path_in_repo=os.path.relpath(file_path, "./ctranslate2_model"),
repo_id=f"mherrador/{model_name}",
token=hf_token
)
# Upload the original model files
for file in os.listdir("./checkpoint"):
file_path = os.path.join("./checkpoint", file)
api.upload_file(
path_or_fileobj=file_path,
path_in_repo=file,
repo_id=f"mherrador/{model_name}",
token=hf_token
)
print("Model trained, converted to CTranslate2 format, and pushed to Hugging Face successfully!")