Download mistral_ft.py from 1earner1/llm_for_Code: direct link, hf CLI and curl.
- Browser
- Download file 7.81 kB
-
https://huggingface.co/1earner1/llm_for_Code/resolve/main/mistral_ft.py
- Command line
-
hf download hf://1earner1/llm_for_Code/mistral_ft.py
-
curl -L -o mistral_ft.py https://huggingface.co/1earner1/llm_for_Code/resolve/main/mistral_ft.py
7.81 kB
| model_name = "mistralai/Mistral-7B-Instruct-v0.3" | |
| #model_name = "bigcode/starcoder2-7b" | |
| #model_name = "dorkai/codeX-1.0" #"Alibaba-NLP/gte-Qwen1.5-7B-instruct" #"google/flan-t5-small" #"microsoft/Phi-3-medium-128k-instruct" #"google/gemma-2-9b-it" # "meta-llama/CodeLlama-7b-hf" #"deepseek-ai/DeepSeek-Coder-V2-Instruct" #"deepseek-ai/DeepSeek-Coder-V2-Lite-Instruct" | |
| out_name = "HPC_2_mistral_iffp_20k_3" #"meta-llama/Meta-Llama-3-8B" #"tiiuae/falcon-40b" #"Phind/Phind-CodeLlama-34B-v2" # "deepseek-ai/DeepSeek-Coder-V2-Instruct" # | |
| from datasets import load_dataset, Dataset | |
| import pandas as pd | |
| import json | |
| import traceback | |
| import torch | |
| # import json | |
| # filepath = "/kaggle/input/code-sim-try1/mutated_graph_all_lang_eq.json" | |
| # examples = [] | |
| # with open(filepath, 'r') as file: | |
| # for l in file: | |
| # examples.append(json.loads(l)) | |
| # len(examples) | |
| # examples[0] | |
| # instruct_tune_dataset = load_dataset("mosaicml/instruct-v3",cache_dir = "/scratch/scai/mtech/aib222688/HF") | |
| # instruct_tune_dataset = instruct_tune_dataset.filter(lambda x: x["source"] == "dolly_hhrlhf") | |
| traindataset_file = "./dataset_lfs/allpairs_data_large.json" | |
| valdataset_file = "./dataset_lfs/allpairs_data_val.json" | |
| testdataset_file = "./llm_for_code/datasets/codecontests/verified_iffp_900.json" | |
| traindata = [] | |
| with open(traindataset_file, 'r') as file: | |
| for l in file: | |
| traindata.append(json.loads(l)) | |
| valdata = [] | |
| with open(valdataset_file, 'r') as file: | |
| for l in file: | |
| valdata.append(json.loads(l)) | |
| testdata = [] | |
| with open(testdataset_file, 'r') as file: | |
| for l in file: | |
| testdata.append(json.loads(l)) | |
| print(len(traindata), len(valdata), len(testdata)) | |
| def create_prompt(sample): | |
| bos_token = "<s>" | |
| original_system_message = "Below is an instruction that describes a task. Write a response that appropriately completes the request." | |
| system_message = "Use the provided input to create an instruction that could have been used to generate the response with an LLM." | |
| #print(sample['prompt']) | |
| response = sample["prompt"].replace(original_system_message, "").replace("\n\n### Instruction\n", "").replace("\n### Response\n", "").strip() | |
| input = sample["response"] | |
| eos_token = "</s>" | |
| full_prompt = "" | |
| full_prompt += bos_token | |
| full_prompt += "### Instruction:" | |
| full_prompt += "\n" + system_message | |
| full_prompt += "\n\n### Input:" | |
| full_prompt += "\n" + input | |
| full_prompt += "\n\n### Response:" | |
| full_prompt += "\n" + response | |
| full_prompt += eos_token | |
| return full_prompt | |
| file_path = './prompt_file.txt' | |
| file_content = "" | |
| with open(file_path, 'r') as file: | |
| file_content = file.read() | |
| def create_prompt(pair): | |
| bos_token = "<s>" | |
| eos_token = "</s>" | |
| if pair['prog1']['probid'] == pair['prog2']['probid']: | |
| response = "Yes" | |
| else: | |
| response = "No" | |
| full_prompt = "" | |
| full_prompt += bos_token | |
| full_prompt += file_content + pair['prog1']['scode'] + "\nProgram 2:" | |
| full_prompt += pair['prog2']['scode']+ "\n### Response:" | |
| full_prompt += "\n" + response | |
| full_prompt += eos_token | |
| return full_prompt | |
| traindataset = Dataset.from_pandas(pd.DataFrame(traindata)) | |
| valdataset = Dataset.from_pandas(pd.DataFrame(valdata)) | |
| testdataset = Dataset.from_pandas(pd.DataFrame(testdata)) | |
| instruct_tune_dataset = {"train": traindataset, | |
| "val" : valdataset, | |
| "test" : testdataset} | |
| print(create_prompt(instruct_tune_dataset["train"][1])) | |
| import os | |
| import torch | |
| from datasets import load_dataset | |
| from transformers import ( | |
| AutoModelForCausalLM, | |
| AutoTokenizer, | |
| BitsAndBytesConfig, | |
| HfArgumentParser, | |
| TrainingArguments, | |
| pipeline, | |
| logging, | |
| ) | |
| from peft import LoraConfig, PeftModel | |
| from trl import SFTTrainer | |
| nf4_config = BitsAndBytesConfig( | |
| load_in_4bit=True, | |
| bnb_4bit_quant_type="nf4", | |
| bnb_4bit_use_double_quant=True, | |
| bnb_4bit_compute_dtype=torch.bfloat16 | |
| ) | |
| model = AutoModelForCausalLM.from_pretrained( | |
| model_name, | |
| device_map='auto', | |
| quantization_config=nf4_config, | |
| use_cache=True, | |
| cache_dir = "/scratch/scai/mtech/aib222688/HF", | |
| #trust_remote_code=True, | |
| ) | |
| tokenizer = AutoTokenizer.from_pretrained(model_name) | |
| tokenizer.pad_token = tokenizer.eos_token | |
| tokenizer.padding_side = "right" | |
| peft_config = LoraConfig( | |
| lora_alpha=16, | |
| lora_dropout=0.1, | |
| r=64, | |
| bias="none", | |
| task_type="CAUSAL_LM" | |
| ) | |
| import peft | |
| model = peft.prepare_model_for_kbit_training(model) | |
| model = peft.get_peft_model(model, peft_config) | |
| args = TrainingArguments( | |
| output_dir = "./"+out_name+"/", | |
| num_train_epochs=4, | |
| #max_steps = 10, | |
| per_device_train_batch_size = 4, | |
| warmup_steps = 0.03, | |
| logging_steps=10, | |
| #save_strategy="epoch", | |
| #evaluation_strategy="epoch", | |
| evaluation_strategy="steps", | |
| eval_steps=100, | |
| save_steps=500, | |
| learning_rate=2e-4, | |
| bf16=True, | |
| lr_scheduler_type='constant', | |
| ) | |
| max_seq_length = 2048 | |
| trainer = SFTTrainer( | |
| model=model, | |
| peft_config=peft_config, | |
| max_seq_length=max_seq_length, | |
| tokenizer=tokenizer, | |
| packing=True, | |
| formatting_func=create_prompt, | |
| args=args, | |
| train_dataset=instruct_tune_dataset["train"], | |
| eval_dataset=instruct_tune_dataset["val"] | |
| ) | |
| # predictions, labels, _ = trainer.predict(instruct_tune_dataset["train"][1]) | |
| # # Convert predictions and labels to text | |
| # predicted_texts = tokenizer.batch_decode(predictions.argmax(axis=-1), skip_special_tokens=True) | |
| # true_texts = tokenizer.batch_decode(labels, skip_special_tokens=True) | |
| # print(f" Pred {predicted_texts}, True {true_texts}") | |
| import time | |
| start = time.time() | |
| try: | |
| trainer.train() | |
| except Exception as e: | |
| print(f"ERRORR\n {e}") | |
| traceback.print_exc() | |
| print(time.time()- start) | |
| def generate_response(prompt, model): | |
| encoded_input = tokenizer(prompt, return_tensors="pt", add_special_tokens=True) | |
| model_inputs = encoded_input.to('cuda') | |
| generated_ids = model.generate(**model_inputs, max_new_tokens=1000, do_sample=True, pad_token_id=tokenizer.eos_token_id) | |
| decoded_output = tokenizer.batch_decode(generated_ids) | |
| return decoded_output[0].replace(prompt, "") | |
| try: | |
| print(generate_response("### Instruction:\nUse the provided input to create an instruction that could have been used to generate the response with an LLM.### Input:\nThere are more than 12,000 species of grass. The most common is Kentucky Bluegrass, because it grows quickly, easily, and is soft to the touch. Rygrass is shiny and bright green colored. Fescues are dark green and shiny. Bermuda grass is harder but can grow in drier soil.\n\n### Response:", trainer.model)) | |
| except Exception as e: | |
| print(f"ERRORR\n {e}") | |
| traceback.print_exc() | |
| try: | |
| trainer.save_model("./"+out_name) | |
| except Exception as e: | |
| print(f"ERRORR\n {e}") | |
| traceback.print_exc() | |
| try: | |
| torch.save(model.state_dict(), "./"+out_name+"/"+'model_state_dict.pkl') | |
| except Exception as e: | |
| print(f"ERRORR\n {e}") | |
| traceback.print_exc() | |
| try: | |
| tester = SFTTrainer( | |
| model=trainer.model, | |
| peft_config=peft_config, | |
| max_seq_length=max_seq_length, | |
| tokenizer=tokenizer, | |
| packing=True, | |
| formatting_func=create_prompt, | |
| args=args, | |
| train_dataset=instruct_tune_dataset["train"], | |
| eval_dataset=instruct_tune_dataset["test"] | |
| ) | |
| print(tester.evaluate()) | |
| except Exception as e: | |
| print(f"ERRORR\n {e}") | |
| traceback.print_exc() | |
| trainer.push_to_hub("FT/mistral-instruct-generation") |