Instructions to use 40Hz/autoresearch-coding-v1 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use 40Hz/autoresearch-coding-v1 with Transformers:
# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("40Hz/autoresearch-coding-v1", device_map="auto") - Notebooks
- Google Colab
- Kaggle
File size: 2,105 Bytes
551ba2d | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 | import os
import torch
from datasets import load_dataset
from peft import LoraConfig
from trl import SFTTrainer, SFTConfig
from transformers import BitsAndBytesConfig
DATA_ID = os.environ.get("CODING_DATA", "theblackcat102/evol-codealpaca-v1")
N_ROWS = int(os.environ.get("CODING_ROWS", "60000"))
dataset = load_dataset(DATA_ID, split="train")
ds = dataset.shuffle(seed=42).select(range(N_ROWS))
ds = ds.train_test_split(test_size=0.05, seed=42)
def fmt(rows):
texts = []
for i, o in zip(rows["instruction"], rows["output"]):
texts.append(f"### Instruction\n{i}\n\n### Response\n{o}<|endoftext|>")
return {"text": texts}
train_ds = ds["train"].map(fmt, batched=True, remove_columns=ds["train"].column_names)
eval_ds = ds["test"].map(fmt, batched=True, remove_columns=ds["test"].column_names)
bnb = BitsAndBytesConfig(
load_in_4bit=True, bnb_4bit_quant_type="nf4",
bnb_4bit_use_double_quant=True, bnb_4bit_compute_dtype=torch.float16,
)
trainer = SFTTrainer(
model="Qwen/Qwen2.5-Coder-0.5B",
train_dataset=train_ds,
eval_dataset=eval_ds,
peft_config=LoraConfig(r=16, lora_alpha=32, target_modules=["q_proj","k_proj","v_proj","o_proj"], lora_dropout=0.05),
args=SFTConfig(
output_dir="autoresearch-coding-v1",
push_to_hub=True,
hub_model_id="40Hz/autoresearch-coding-v1",
num_train_epochs=2,
per_device_train_batch_size=4,
gradient_accumulation_steps=4,
gradient_checkpointing=True,
eval_strategy="steps",
eval_steps=200,
logging_steps=20,
save_steps=200,
save_total_limit=3,
fp16=True,
learning_rate=2e-4,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
optim="paged_adamw_8bit",
max_length=1024,
report_to="trackio",
trackio_space_id=os.environ.get("TRACKIO_SPACE_ID", "40Hz/autoresearch-gpu"),
model_init_kwargs={"quantization_config": bnb, "torch_dtype": torch.float16},
),
)
trainer.train()
trainer.push_to_hub()
print("DONE coding-v1 pushed to 40Hz/autoresearch-coding-v1") |