File size: 2,801 Bytes
bc6c786
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
{
  "paths": {
    "data_dir": "data",
    "raw_dir": "data/raw",
    "processed_dir": "data/processed",
    "train_dir": "data/train",
    "validation_dir": "data/validation",
    "test_dir": "data/test",
    "reports_dir": "reports",
    "outputs_dir": "outputs"
  },
  "data": {
    "nl2bash_repo": "jiacheng-ye/nl2bash",
    "linux_commands_repo": "mecha-org/linux-command-dataset",
    "nl2bash_archive_url": "https://www.dropbox.com/s/wy7uahzbir7lrq1/nl2bash.zip?dl=1",
    "min_instruction_chars": 5,
    "min_command_chars": 2,
    "max_instruction_chars": 600,
    "max_command_chars": 600,
    "head_command_cap": 2500,
    "exact_command_cap": 40,
    "near_duplicate_threshold": 0.9,
    "drop_near_duplicates": false,
    "split_grouping": "template",
    "train_ratio": 0.9,
    "validation_ratio": 0.05,
    "test_ratio": 0.05,
    "seed": 42
  },
  "model": {
    "model_name": "LiquidAI/LFM2-700M",
    "dtype": "bfloat16",
    "attn_implementation": "sdpa",
    "system_prompt": null,
    "max_length": 128,
    "trust_remote_code": false
  },
  "lora": {
    "r": 32,
    "alpha": 64,
    "dropout": 0.05,
    "target_modules": [
      "q_proj",
      "k_proj",
      "v_proj",
      "out_proj",
      "w1",
      "w2",
      "w3",
      "in_proj"
    ],
    "exclude_modules": [
      "conv.conv"
    ],
    "modules_to_save": []
  },
  "training": {
    "method": "full",
    "output_dir": "outputs/lfm2-linux-command",
    "run_name": "lfm2-700m-linux-command",
    "num_train_epochs": 3,
    "per_device_train_batch_size": 16,
    "per_device_eval_batch_size": 32,
    "gradient_accumulation_steps": 2,
    "learning_rate": 3e-05,
    "lr_scheduler_type": "cosine",
    "warmup_ratio": 0.03,
    "weight_decay": 0.01,
    "max_grad_norm": 1.0,
    "optim": "adamw_torch_fused",
    "adam_beta1": 0.9,
    "adam_beta2": 0.95,
    "adam_epsilon": 1e-08,
    "bf16": true,
    "fp16": false,
    "gradient_checkpointing": false,
    "group_by_length": true,
    "logging_steps": 20,
    "eval_strategy": "epoch",
    "save_strategy": "epoch",
    "save_total_limit": 2,
    "load_best_model_at_end": true,
    "metric_for_best_model": "eval_loss",
    "greater_is_better": false,
    "early_stopping_patience": 2,
    "dataloader_num_workers": 4,
    "seed": 42,
    "report_to": []
  },
  "generation": {
    "max_new_tokens": 128,
    "do_sample": false,
    "temperature": 1.0,
    "top_p": 1.0,
    "num_beams": 1,
    "repetition_penalty": 1.0,
    "batch_size": 32
  },
  "evaluation": {
    "limit": null,
    "execute_in_sandbox": false,
    "sandbox_image": "linux-cmd-sandbox:latest",
    "sandbox_timeout_s": 10,
    "sandbox_memory": "256m",
    "sandbox_cpus": "1.0",
    "sandbox_pids_limit": 128,
    "sandbox_network": false,
    "max_execution_examples": 200
  }
}