model_name = "mistralai/Mistral-7B-Instruct-v0.3" #model_name = "bigcode/starcoder2-7b" #model_name = "dorkai/codeX-1.0" #"Alibaba-NLP/gte-Qwen1.5-7B-instruct" #"google/flan-t5-small" #"microsoft/Phi-3-medium-128k-instruct" #"google/gemma-2-9b-it" # "meta-llama/CodeLlama-7b-hf" #"deepseek-ai/DeepSeek-Coder-V2-Instruct" #"deepseek-ai/DeepSeek-Coder-V2-Lite-Instruct" out_name = "HPC_2_mistral_iffp_20k_3" #"meta-llama/Meta-Llama-3-8B" #"tiiuae/falcon-40b" #"Phind/Phind-CodeLlama-34B-v2" # "deepseek-ai/DeepSeek-Coder-V2-Instruct" # from datasets import load_dataset, Dataset import pandas as pd import json import traceback import torch # import json # filepath = "/kaggle/input/code-sim-try1/mutated_graph_all_lang_eq.json" # examples = [] # with open(filepath, 'r') as file: # for l in file: # examples.append(json.loads(l)) # len(examples) # examples[0] # instruct_tune_dataset = load_dataset("mosaicml/instruct-v3",cache_dir = "/scratch/scai/mtech/aib222688/HF") # instruct_tune_dataset = instruct_tune_dataset.filter(lambda x: x["source"] == "dolly_hhrlhf") traindataset_file = "./dataset_lfs/allpairs_data_large.json" valdataset_file = "./dataset_lfs/allpairs_data_val.json" testdataset_file = "./llm_for_code/datasets/codecontests/verified_iffp_900.json" traindata = [] with open(traindataset_file, 'r') as file: for l in file: traindata.append(json.loads(l)) valdata = [] with open(valdataset_file, 'r') as file: for l in file: valdata.append(json.loads(l)) testdata = [] with open(testdataset_file, 'r') as file: for l in file: testdata.append(json.loads(l)) print(len(traindata), len(valdata), len(testdata)) def create_prompt(sample): bos_token = "" original_system_message = "Below is an instruction that describes a task. Write a response that appropriately completes the request." system_message = "Use the provided input to create an instruction that could have been used to generate the response with an LLM." #print(sample['prompt']) response = sample["prompt"].replace(original_system_message, "").replace("\n\n### Instruction\n", "").replace("\n### Response\n", "").strip() input = sample["response"] eos_token = "" full_prompt = "" full_prompt += bos_token full_prompt += "### Instruction:" full_prompt += "\n" + system_message full_prompt += "\n\n### Input:" full_prompt += "\n" + input full_prompt += "\n\n### Response:" full_prompt += "\n" + response full_prompt += eos_token return full_prompt file_path = './prompt_file.txt' file_content = "" with open(file_path, 'r') as file: file_content = file.read() def create_prompt(pair): bos_token = "" eos_token = "" if pair['prog1']['probid'] == pair['prog2']['probid']: response = "Yes" else: response = "No" full_prompt = "" full_prompt += bos_token full_prompt += file_content + pair['prog1']['scode'] + "\nProgram 2:" full_prompt += pair['prog2']['scode']+ "\n### Response:" full_prompt += "\n" + response full_prompt += eos_token return full_prompt traindataset = Dataset.from_pandas(pd.DataFrame(traindata)) valdataset = Dataset.from_pandas(pd.DataFrame(valdata)) testdataset = Dataset.from_pandas(pd.DataFrame(testdata)) instruct_tune_dataset = {"train": traindataset, "val" : valdataset, "test" : testdataset} print(create_prompt(instruct_tune_dataset["train"][1])) import os import torch from datasets import load_dataset from transformers import ( AutoModelForCausalLM, AutoTokenizer, BitsAndBytesConfig, HfArgumentParser, TrainingArguments, pipeline, logging, ) from peft import LoraConfig, PeftModel from trl import SFTTrainer nf4_config = BitsAndBytesConfig( load_in_4bit=True, bnb_4bit_quant_type="nf4", bnb_4bit_use_double_quant=True, bnb_4bit_compute_dtype=torch.bfloat16 ) model = AutoModelForCausalLM.from_pretrained( model_name, device_map='auto', quantization_config=nf4_config, use_cache=True, cache_dir = "/scratch/scai/mtech/aib222688/HF", #trust_remote_code=True, ) tokenizer = AutoTokenizer.from_pretrained(model_name) tokenizer.pad_token = tokenizer.eos_token tokenizer.padding_side = "right" peft_config = LoraConfig( lora_alpha=16, lora_dropout=0.1, r=64, bias="none", task_type="CAUSAL_LM" ) import peft model = peft.prepare_model_for_kbit_training(model) model = peft.get_peft_model(model, peft_config) args = TrainingArguments( output_dir = "./"+out_name+"/", num_train_epochs=4, #max_steps = 10, per_device_train_batch_size = 4, warmup_steps = 0.03, logging_steps=10, #save_strategy="epoch", #evaluation_strategy="epoch", evaluation_strategy="steps", eval_steps=100, save_steps=500, learning_rate=2e-4, bf16=True, lr_scheduler_type='constant', ) max_seq_length = 2048 trainer = SFTTrainer( model=model, peft_config=peft_config, max_seq_length=max_seq_length, tokenizer=tokenizer, packing=True, formatting_func=create_prompt, args=args, train_dataset=instruct_tune_dataset["train"], eval_dataset=instruct_tune_dataset["val"] ) # predictions, labels, _ = trainer.predict(instruct_tune_dataset["train"][1]) # # Convert predictions and labels to text # predicted_texts = tokenizer.batch_decode(predictions.argmax(axis=-1), skip_special_tokens=True) # true_texts = tokenizer.batch_decode(labels, skip_special_tokens=True) # print(f" Pred {predicted_texts}, True {true_texts}") import time start = time.time() try: trainer.train() except Exception as e: print(f"ERRORR\n {e}") traceback.print_exc() print(time.time()- start) def generate_response(prompt, model): encoded_input = tokenizer(prompt, return_tensors="pt", add_special_tokens=True) model_inputs = encoded_input.to('cuda') generated_ids = model.generate(**model_inputs, max_new_tokens=1000, do_sample=True, pad_token_id=tokenizer.eos_token_id) decoded_output = tokenizer.batch_decode(generated_ids) return decoded_output[0].replace(prompt, "") try: print(generate_response("### Instruction:\nUse the provided input to create an instruction that could have been used to generate the response with an LLM.### Input:\nThere are more than 12,000 species of grass. The most common is Kentucky Bluegrass, because it grows quickly, easily, and is soft to the touch. Rygrass is shiny and bright green colored. Fescues are dark green and shiny. Bermuda grass is harder but can grow in drier soil.\n\n### Response:", trainer.model)) except Exception as e: print(f"ERRORR\n {e}") traceback.print_exc() try: trainer.save_model("./"+out_name) except Exception as e: print(f"ERRORR\n {e}") traceback.print_exc() try: torch.save(model.state_dict(), "./"+out_name+"/"+'model_state_dict.pkl') except Exception as e: print(f"ERRORR\n {e}") traceback.print_exc() try: tester = SFTTrainer( model=trainer.model, peft_config=peft_config, max_seq_length=max_seq_length, tokenizer=tokenizer, packing=True, formatting_func=create_prompt, args=args, train_dataset=instruct_tune_dataset["train"], eval_dataset=instruct_tune_dataset["test"] ) print(tester.evaluate()) except Exception as e: print(f"ERRORR\n {e}") traceback.print_exc() trainer.push_to_hub("FT/mistral-instruct-generation")