gpt2sin / dataload.py
dastharak's picture
Upload 4 files
9a9467f verified
Raw History Blame Contribute Delete
9.7 kB
import pandas as pd
import torch
from torch.utils.data import Dataset, DataLoader
from transformers import AutoTokenizer, PreTrainedTokenizer
import os
# --- 1. CONFIGURATION AND FILE LOADING ---
# Replace 'path/to/your/converted_data.csv' with the actual path
FILE_PATH = 'converted_data.csv'
KAGGLE_PATH = '/kaggle/'
MAX_LENGTH = 128 # Maximum sequence length for the model
def load_data(this_dir_path,file_path=FILE_PATH):
"""Loads the CSV, selecting Sinhala as source and English as target."""
print(f"Loading data from: {this_dir_path}{file_path}")
f_path = this_dir_path+file_path
print(f"f_path:{f_path}")
if os.path.exists(f_path):
print(f"exists?{f_path}")
df = pd.read_csv(f_path, usecols=['Sinhala', 'English'])
print(f"df:{len(df)}")
else:
print("Note: File not found. Using dummy data for demonstration.")
data = {
'Sinhala': [
"ඔබට කොහොමද?",
"හෙට මම එනවා.",
"කරුණාකර උදව් කරන්න."
],
'English': [
"How are you?",
"I will come tomorrow.",
"Please help."
],
'Singlish': [
"Oya kohomada?",
"Heta mama enawa.",
"Karunakara udaw karanna."
]
}
df = pd.DataFrame(data, columns=['Sinhala', 'English', 'Singlish'])
# Select the columns you need for the task
df = df[['Sinhala', 'English']]
print(f"Loaded {len(df)} samples.")
return df
# --- 2. TOKENIZER SETUP ---
def setup_tokenizer():
"""
Initializes a GPT-2 tokenizer and adds special tokens for Seq2S
eq.
NOTE ON SINHALA:
GPT-2 was primarily trained on English text. For better Sinhala support,
you would ideally use a multilingual tokenizer (like XLM-R or mBART)
or train a custom tokenizer using your entire dataset.
For this preliminary code, we use a standard GPT2Tokenizer and add the
necessary control tokens.
"""
# Using 'gpt2' base tokenizer
tokenizer = AutoTokenizer.from_pretrained("gpt2")
# Define custom tokens needed for sequence-to-sequence:
# <SEP> : Separator between the Source (Sinhala) and Target (English) sentence
# <PAD> : Padding token (GPT2 doesn't have one by default, crucial for batching)
new_tokens = {
'pad_token': '<PAD>',
'sep_token': '<SEP>',
}
# Add new tokens and resize the tokenizer vocabulary
num_added_toks = tokenizer.add_special_tokens(new_tokens)
# Set the padding side to 'left' or 'right'.
# For decoder-only models (like GPT-2), 'left' padding is often preferred
# for faster attention, but 'right' is also common. We'll use 'right'.
tokenizer.padding_side = "right"
print(f"Added {num_added_toks} custom tokens.")
print(f"BOS Token: {tokenizer.bos_token} ({tokenizer.bos_token_id})")
print(f"SEP Token: {tokenizer.sep_token} ({tokenizer.sep_token_id})")
return tokenizer
# --- 3. PYTORCH CUSTOM DATASET ---
class TranslationDataset(Dataset):
"""
Custom Dataset to prepare data for a decoder-only model (GPT-2)
used in a sequence-to-sequence (translation) task.
"""
def __init__(self, data_frame: pd.DataFrame, tokenizer: PreTrainedTokenizer, max_length: int):
self.tokenizer = tokenizer
self.data = data_frame
self.max_length = max_length
def __len__(self):
return len(self.data)
def __getitem__(self, idx):
# 1. Retrieve the Source (Sinhala) and Target (English) sentences
source_text = self.data.iloc[idx]['Sinhala']
target_text = self.data.iloc[idx]['English']
# 2. Construct the single sequence for CLM (Causal Language Modeling)
# Format: <BOS> Sinhala_Sentence <SEP> English_Sentence <EOS>
full_sequence = (
self.tokenizer.bos_token +
source_text +
self.tokenizer.sep_token +
target_text +
self.tokenizer.eos_token
)
# 3. Tokenize and truncate
tokenized_sequence = self.tokenizer(
full_sequence,
max_length=self.max_length,
truncation=True,
return_tensors='pt' # Return PyTorch tensors
)
# Squeeze to remove the batch dimension (which is 1 here)
input_ids = tokenized_sequence['input_ids'].squeeze(0)
attention_mask = tokenized_sequence['attention_mask'].squeeze(0)
# 4. Prepare Labels for CLM Loss (Shifted Input)
# In PyTorch's GPT2 implementation, the model shifts the labels internally.
# We just need to pass the input_ids as labels.
# labels = input_ids.clone()
# --- IMPORTANT: Loss Masking for Translation ---
# We only want the model to calculate loss over the GENERATED tokens
# (the English/target part).
# PyTorch's CrossEntropyLoss ignores targets with a value of -100.
# Find the index of the <SEP> token (start of the target sequence)
sep_token_id = self.tokenizer.sep_token_id
sep_index = (input_ids == sep_token_id).nonzero(as_tuple=True)[0]
# Check if the SEP token exists (it should, unless truncated out)
if sep_index.numel() > 0:
# Mask out the Source part and the <SEP> token itself
# Source tokens, <BOS>, and <SEP> are set to -100
mask_end_index = sep_index[0].item()
# Initialize labels as a copy of input_ids
labels = input_ids.clone()
# Set the Source sentence tokens (including <BOS> and <SEP>) to -100
labels[:mask_end_index + 1] = -100
else:
# If <SEP> is not found (due to truncation), mask the whole sequence
labels = torch.full_like(input_ids, -100)
return {
'input_ids': input_ids,
'attention_mask': attention_mask,
'labels': labels
}
# --- 4. DATA COLLATOR AND DATALOADER SETUP ---
def data_collator_fn(batch_list, pad_token_id):
"""
Custom collate function to handle padding for the batch.
"""
# Stack the tensors for padding
input_ids = [item['input_ids'] for item in batch_list]
attention_mask = [item['attention_mask'] for item in batch_list]
labels = [item['labels'] for item in batch_list]
# Padding function
# NOTE: Since we set padding_side="right" in the tokenizer,
# the standard pad_sequence handles this correctly.
input_ids_padded = torch.nn.utils.rnn.pad_sequence(
input_ids, batch_first=True, padding_value=pad_token_id
)
attention_mask_padded = torch.nn.utils.rnn.pad_sequence(
attention_mask, batch_first=True, padding_value=0 # Attention mask uses 0 for padding
)
labels_padded = torch.nn.utils.rnn.pad_sequence(
labels, batch_first=True, padding_value=-100 # Labels use -100 for padding/masking
)
return {
'input_ids': input_ids_padded,
'attention_mask': attention_mask_padded,
'labels': labels_padded
}
# --- MAIN EXECUTION BLOCK ---
if __name__ == '__main__':
# 1. Load Data
data_df = load_data(FILE_PATH)
# 2. Setup Tokenizer
tokenizer = setup_tokenizer()
# 3. Create Dataset
translation_dataset = TranslationDataset(data_df, tokenizer, MAX_LENGTH)
# 4. Create DataLoader
BATCH_SIZE = 4
# Pass a lambda function to the collate_fn argument,
# which binds the tokenizer's pad_token_id
train_dataloader = DataLoader(
translation_dataset,
batch_size=BATCH_SIZE,
shuffle=True,
collate_fn=lambda batch_list: data_collator_fn(batch_list, tokenizer.pad_token_id)
)
print("\n--- DataLoader Example ---")
print(f"Total batches: {len(train_dataloader)}")
# 5. Inspect a single batch
for batch in train_dataloader:
print(f"\nBatch 'input_ids' shape: {batch['input_ids'].shape}")
print(f"Batch 'labels' shape: {batch['labels'].shape}")
# Print a sample from the batch to verify masking
sample_index = 0
print("\n--- Sample 1 Verification ---")
# Decode the full input sequence
print("Input Sequence (Decoded):")
print(tokenizer.decode(batch['input_ids'][sample_index], skip_special_tokens=False))
# Find all tokens that are NOT masked (-100)
target_tokens = batch['labels'][sample_index].clone()
target_tokens[target_tokens == -100] = tokenizer.pad_token_id # replace -100 with PAD for decoding
print("\nTarget Labels (Decoded - Only English should be visible):")
print(tokenizer.decode(target_tokens, skip_special_tokens=True))
print("\nRaw Labels Tensor (Verifying Masking):")
# Show a snippet of the labels tensor to confirm -100 for the source part
print(batch['labels'][sample_index])
break # Stop after inspecting the first batch
# You can now iterate over `train_dataloader` to get batches for training.
if __name__ == '__main__':
df = load_data(FILE_PATH) # data frame
df.head()