Download dataload.py from dastharak/gpt2sin: direct link, hf CLI and curl.
- Browser
- Download file 9.7 kB
-
https://huggingface.co/dastharak/gpt2sin/resolve/main/dataload.py
- Command line
-
hf download hf://dastharak/gpt2sin/dataload.py
-
curl -L -o dataload.py https://huggingface.co/dastharak/gpt2sin/resolve/main/dataload.py
9.7 kB
| import pandas as pd | |
| import torch | |
| from torch.utils.data import Dataset, DataLoader | |
| from transformers import AutoTokenizer, PreTrainedTokenizer | |
| import os | |
| # --- 1. CONFIGURATION AND FILE LOADING --- | |
| # Replace 'path/to/your/converted_data.csv' with the actual path | |
| FILE_PATH = 'converted_data.csv' | |
| KAGGLE_PATH = '/kaggle/' | |
| MAX_LENGTH = 128 # Maximum sequence length for the model | |
| def load_data(this_dir_path,file_path=FILE_PATH): | |
| """Loads the CSV, selecting Sinhala as source and English as target.""" | |
| print(f"Loading data from: {this_dir_path}{file_path}") | |
| f_path = this_dir_path+file_path | |
| print(f"f_path:{f_path}") | |
| if os.path.exists(f_path): | |
| print(f"exists?{f_path}") | |
| df = pd.read_csv(f_path, usecols=['Sinhala', 'English']) | |
| print(f"df:{len(df)}") | |
| else: | |
| print("Note: File not found. Using dummy data for demonstration.") | |
| data = { | |
| 'Sinhala': [ | |
| "ඔබට කොහොමද?", | |
| "හෙට මම එනවා.", | |
| "කරුණාකර උදව් කරන්න." | |
| ], | |
| 'English': [ | |
| "How are you?", | |
| "I will come tomorrow.", | |
| "Please help." | |
| ], | |
| 'Singlish': [ | |
| "Oya kohomada?", | |
| "Heta mama enawa.", | |
| "Karunakara udaw karanna." | |
| ] | |
| } | |
| df = pd.DataFrame(data, columns=['Sinhala', 'English', 'Singlish']) | |
| # Select the columns you need for the task | |
| df = df[['Sinhala', 'English']] | |
| print(f"Loaded {len(df)} samples.") | |
| return df | |
| # --- 2. TOKENIZER SETUP --- | |
| def setup_tokenizer(): | |
| """ | |
| Initializes a GPT-2 tokenizer and adds special tokens for Seq2S | |
| eq. | |
| NOTE ON SINHALA: | |
| GPT-2 was primarily trained on English text. For better Sinhala support, | |
| you would ideally use a multilingual tokenizer (like XLM-R or mBART) | |
| or train a custom tokenizer using your entire dataset. | |
| For this preliminary code, we use a standard GPT2Tokenizer and add the | |
| necessary control tokens. | |
| """ | |
| # Using 'gpt2' base tokenizer | |
| tokenizer = AutoTokenizer.from_pretrained("gpt2") | |
| # Define custom tokens needed for sequence-to-sequence: | |
| # <SEP> : Separator between the Source (Sinhala) and Target (English) sentence | |
| # <PAD> : Padding token (GPT2 doesn't have one by default, crucial for batching) | |
| new_tokens = { | |
| 'pad_token': '<PAD>', | |
| 'sep_token': '<SEP>', | |
| } | |
| # Add new tokens and resize the tokenizer vocabulary | |
| num_added_toks = tokenizer.add_special_tokens(new_tokens) | |
| # Set the padding side to 'left' or 'right'. | |
| # For decoder-only models (like GPT-2), 'left' padding is often preferred | |
| # for faster attention, but 'right' is also common. We'll use 'right'. | |
| tokenizer.padding_side = "right" | |
| print(f"Added {num_added_toks} custom tokens.") | |
| print(f"BOS Token: {tokenizer.bos_token} ({tokenizer.bos_token_id})") | |
| print(f"SEP Token: {tokenizer.sep_token} ({tokenizer.sep_token_id})") | |
| return tokenizer | |
| # --- 3. PYTORCH CUSTOM DATASET --- | |
| class TranslationDataset(Dataset): | |
| """ | |
| Custom Dataset to prepare data for a decoder-only model (GPT-2) | |
| used in a sequence-to-sequence (translation) task. | |
| """ | |
| def __init__(self, data_frame: pd.DataFrame, tokenizer: PreTrainedTokenizer, max_length: int): | |
| self.tokenizer = tokenizer | |
| self.data = data_frame | |
| self.max_length = max_length | |
| def __len__(self): | |
| return len(self.data) | |
| def __getitem__(self, idx): | |
| # 1. Retrieve the Source (Sinhala) and Target (English) sentences | |
| source_text = self.data.iloc[idx]['Sinhala'] | |
| target_text = self.data.iloc[idx]['English'] | |
| # 2. Construct the single sequence for CLM (Causal Language Modeling) | |
| # Format: <BOS> Sinhala_Sentence <SEP> English_Sentence <EOS> | |
| full_sequence = ( | |
| self.tokenizer.bos_token + | |
| source_text + | |
| self.tokenizer.sep_token + | |
| target_text + | |
| self.tokenizer.eos_token | |
| ) | |
| # 3. Tokenize and truncate | |
| tokenized_sequence = self.tokenizer( | |
| full_sequence, | |
| max_length=self.max_length, | |
| truncation=True, | |
| return_tensors='pt' # Return PyTorch tensors | |
| ) | |
| # Squeeze to remove the batch dimension (which is 1 here) | |
| input_ids = tokenized_sequence['input_ids'].squeeze(0) | |
| attention_mask = tokenized_sequence['attention_mask'].squeeze(0) | |
| # 4. Prepare Labels for CLM Loss (Shifted Input) | |
| # In PyTorch's GPT2 implementation, the model shifts the labels internally. | |
| # We just need to pass the input_ids as labels. | |
| # labels = input_ids.clone() | |
| # --- IMPORTANT: Loss Masking for Translation --- | |
| # We only want the model to calculate loss over the GENERATED tokens | |
| # (the English/target part). | |
| # PyTorch's CrossEntropyLoss ignores targets with a value of -100. | |
| # Find the index of the <SEP> token (start of the target sequence) | |
| sep_token_id = self.tokenizer.sep_token_id | |
| sep_index = (input_ids == sep_token_id).nonzero(as_tuple=True)[0] | |
| # Check if the SEP token exists (it should, unless truncated out) | |
| if sep_index.numel() > 0: | |
| # Mask out the Source part and the <SEP> token itself | |
| # Source tokens, <BOS>, and <SEP> are set to -100 | |
| mask_end_index = sep_index[0].item() | |
| # Initialize labels as a copy of input_ids | |
| labels = input_ids.clone() | |
| # Set the Source sentence tokens (including <BOS> and <SEP>) to -100 | |
| labels[:mask_end_index + 1] = -100 | |
| else: | |
| # If <SEP> is not found (due to truncation), mask the whole sequence | |
| labels = torch.full_like(input_ids, -100) | |
| return { | |
| 'input_ids': input_ids, | |
| 'attention_mask': attention_mask, | |
| 'labels': labels | |
| } | |
| # --- 4. DATA COLLATOR AND DATALOADER SETUP --- | |
| def data_collator_fn(batch_list, pad_token_id): | |
| """ | |
| Custom collate function to handle padding for the batch. | |
| """ | |
| # Stack the tensors for padding | |
| input_ids = [item['input_ids'] for item in batch_list] | |
| attention_mask = [item['attention_mask'] for item in batch_list] | |
| labels = [item['labels'] for item in batch_list] | |
| # Padding function | |
| # NOTE: Since we set padding_side="right" in the tokenizer, | |
| # the standard pad_sequence handles this correctly. | |
| input_ids_padded = torch.nn.utils.rnn.pad_sequence( | |
| input_ids, batch_first=True, padding_value=pad_token_id | |
| ) | |
| attention_mask_padded = torch.nn.utils.rnn.pad_sequence( | |
| attention_mask, batch_first=True, padding_value=0 # Attention mask uses 0 for padding | |
| ) | |
| labels_padded = torch.nn.utils.rnn.pad_sequence( | |
| labels, batch_first=True, padding_value=-100 # Labels use -100 for padding/masking | |
| ) | |
| return { | |
| 'input_ids': input_ids_padded, | |
| 'attention_mask': attention_mask_padded, | |
| 'labels': labels_padded | |
| } | |
| # --- MAIN EXECUTION BLOCK --- | |
| if __name__ == '__main__': | |
| # 1. Load Data | |
| data_df = load_data(FILE_PATH) | |
| # 2. Setup Tokenizer | |
| tokenizer = setup_tokenizer() | |
| # 3. Create Dataset | |
| translation_dataset = TranslationDataset(data_df, tokenizer, MAX_LENGTH) | |
| # 4. Create DataLoader | |
| BATCH_SIZE = 4 | |
| # Pass a lambda function to the collate_fn argument, | |
| # which binds the tokenizer's pad_token_id | |
| train_dataloader = DataLoader( | |
| translation_dataset, | |
| batch_size=BATCH_SIZE, | |
| shuffle=True, | |
| collate_fn=lambda batch_list: data_collator_fn(batch_list, tokenizer.pad_token_id) | |
| ) | |
| print("\n--- DataLoader Example ---") | |
| print(f"Total batches: {len(train_dataloader)}") | |
| # 5. Inspect a single batch | |
| for batch in train_dataloader: | |
| print(f"\nBatch 'input_ids' shape: {batch['input_ids'].shape}") | |
| print(f"Batch 'labels' shape: {batch['labels'].shape}") | |
| # Print a sample from the batch to verify masking | |
| sample_index = 0 | |
| print("\n--- Sample 1 Verification ---") | |
| # Decode the full input sequence | |
| print("Input Sequence (Decoded):") | |
| print(tokenizer.decode(batch['input_ids'][sample_index], skip_special_tokens=False)) | |
| # Find all tokens that are NOT masked (-100) | |
| target_tokens = batch['labels'][sample_index].clone() | |
| target_tokens[target_tokens == -100] = tokenizer.pad_token_id # replace -100 with PAD for decoding | |
| print("\nTarget Labels (Decoded - Only English should be visible):") | |
| print(tokenizer.decode(target_tokens, skip_special_tokens=True)) | |
| print("\nRaw Labels Tensor (Verifying Masking):") | |
| # Show a snippet of the labels tensor to confirm -100 for the source part | |
| print(batch['labels'][sample_index]) | |
| break # Stop after inspecting the first batch | |
| # You can now iterate over `train_dataloader` to get batches for training. | |
| if __name__ == '__main__': | |
| df = load_data(FILE_PATH) # data frame | |
| df.head() |