Upload folder using huggingface_hub
Browse files- infer/qwen/spider_data/sft__sft_qwen3_0.6b__ckpt1090__test__spider_data_sql_result.json +1 -0
- infer/qwen/spider_data/sft__sft_qwen3_4b__ckpt1090__test__spider_data_sql_result.json +0 -0
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/1090/README.md +2 -2
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/1090/adapter_config.json +6 -6
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/1090/adapter_model.bin +1 -1
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/1090/chat_template.jinja +1 -29
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/1090/tokenizer_config.json +1 -1
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/218/README.md +2 -2
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/218/adapter_config.json +6 -6
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/218/adapter_model.bin +1 -1
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/218/chat_template.jinja +1 -29
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/218/tokenizer_config.json +1 -1
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/436/README.md +2 -2
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/436/adapter_config.json +6 -6
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/436/adapter_model.bin +1 -1
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/436/chat_template.jinja +1 -29
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/436/tokenizer_config.json +1 -1
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/654/README.md +2 -2
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/654/adapter_config.json +6 -6
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/654/adapter_model.bin +1 -1
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/654/chat_template.jinja +1 -29
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/654/tokenizer_config.json +1 -1
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/872/README.md +2 -2
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/872/adapter_config.json +6 -6
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/872/adapter_model.bin +1 -1
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/872/chat_template.jinja +1 -29
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/872/tokenizer_config.json +1 -1
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/args.json +1 -1
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/eval/0/answers.jsonl +0 -0
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/eval/1/answers.jsonl +0 -0
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/eval/2/answers.jsonl +0 -0
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/eval/3/answers.jsonl +0 -0
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/eval/4/answers.jsonl +0 -0
- qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/log.txt +63 -0
infer/qwen/spider_data/sft__sft_qwen3_0.6b__ckpt1090__test__spider_data_sql_result.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
[]
|
infer/qwen/spider_data/sft__sft_qwen3_4b__ckpt1090__test__spider_data_sql_result.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/1090/README.md
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
---
|
| 2 |
-
base_model: Qwen/Qwen3-4B
|
| 3 |
library_name: peft
|
| 4 |
pipeline_tag: text-generation
|
| 5 |
tags:
|
| 6 |
-
- base_model:adapter:Qwen/Qwen3-4B
|
| 7 |
- lora
|
| 8 |
- transformers
|
| 9 |
---
|
|
|
|
| 1 |
---
|
| 2 |
+
base_model: Qwen/Qwen3-4B-Instruct-2507
|
| 3 |
library_name: peft
|
| 4 |
pipeline_tag: text-generation
|
| 5 |
tags:
|
| 6 |
+
- base_model:adapter:Qwen/Qwen3-4B-Instruct-2507
|
| 7 |
- lora
|
| 8 |
- transformers
|
| 9 |
---
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/1090/adapter_config.json
CHANGED
|
@@ -3,7 +3,7 @@
|
|
| 3 |
"alpha_pattern": {},
|
| 4 |
"arrow_config": null,
|
| 5 |
"auto_mapping": null,
|
| 6 |
-
"base_model_name_or_path": "Qwen/Qwen3-4B",
|
| 7 |
"bias": "none",
|
| 8 |
"corda_config": null,
|
| 9 |
"ensure_weight_tying": false,
|
|
@@ -29,13 +29,13 @@
|
|
| 29 |
"rank_pattern": {},
|
| 30 |
"revision": null,
|
| 31 |
"target_modules": [
|
| 32 |
-
"k_proj",
|
| 33 |
-
"q_proj",
|
| 34 |
-
"v_proj",
|
| 35 |
"down_proj",
|
| 36 |
"gate_proj",
|
| 37 |
-
"
|
| 38 |
-
"up_proj"
|
|
|
|
|
|
|
|
|
|
| 39 |
],
|
| 40 |
"target_parameters": null,
|
| 41 |
"task_type": "CAUSAL_LM",
|
|
|
|
| 3 |
"alpha_pattern": {},
|
| 4 |
"arrow_config": null,
|
| 5 |
"auto_mapping": null,
|
| 6 |
+
"base_model_name_or_path": "Qwen/Qwen3-4B-Instruct-2507",
|
| 7 |
"bias": "none",
|
| 8 |
"corda_config": null,
|
| 9 |
"ensure_weight_tying": false,
|
|
|
|
| 29 |
"rank_pattern": {},
|
| 30 |
"revision": null,
|
| 31 |
"target_modules": [
|
|
|
|
|
|
|
|
|
|
| 32 |
"down_proj",
|
| 33 |
"gate_proj",
|
| 34 |
+
"q_proj",
|
| 35 |
+
"up_proj",
|
| 36 |
+
"v_proj",
|
| 37 |
+
"k_proj",
|
| 38 |
+
"o_proj"
|
| 39 |
],
|
| 40 |
"target_parameters": null,
|
| 41 |
"task_type": "CAUSAL_LM",
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/1090/adapter_model.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 132200851
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6465f5c097501c718075fe7928288536990ef072eee9ac049b72c438e2d8277a
|
| 3 |
size 132200851
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/1090/chat_template.jinja
CHANGED
|
@@ -14,14 +14,6 @@
|
|
| 14 |
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
|
| 15 |
{%- endif %}
|
| 16 |
{%- endif %}
|
| 17 |
-
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
|
| 18 |
-
{%- for message in messages[::-1] %}
|
| 19 |
-
{%- set index = (messages|length - 1) - loop.index0 %}
|
| 20 |
-
{%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('<tool_response>') and message.content.endswith('</tool_response>')) %}
|
| 21 |
-
{%- set ns.multi_step_tool = false %}
|
| 22 |
-
{%- set ns.last_query_index = index %}
|
| 23 |
-
{%- endif %}
|
| 24 |
-
{%- endfor %}
|
| 25 |
{%- for message in messages %}
|
| 26 |
{%- if message.content is string %}
|
| 27 |
{%- set content = message.content %}
|
|
@@ -31,24 +23,7 @@
|
|
| 31 |
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
|
| 32 |
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
| 33 |
{%- elif message.role == "assistant" %}
|
| 34 |
-
{
|
| 35 |
-
{%- if message.reasoning_content is string %}
|
| 36 |
-
{%- set reasoning_content = message.reasoning_content %}
|
| 37 |
-
{%- else %}
|
| 38 |
-
{%- if '</think>' in content %}
|
| 39 |
-
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
|
| 40 |
-
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
|
| 41 |
-
{%- endif %}
|
| 42 |
-
{%- endif %}
|
| 43 |
-
{%- if loop.index0 > ns.last_query_index %}
|
| 44 |
-
{%- if loop.last or (not loop.last and reasoning_content) %}
|
| 45 |
-
{{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content.strip('\n') + '\n</think>\n\n' + content.lstrip('\n') }}
|
| 46 |
-
{%- else %}
|
| 47 |
-
{{- '<|im_start|>' + message.role + '\n' + content }}
|
| 48 |
-
{%- endif %}
|
| 49 |
-
{%- else %}
|
| 50 |
-
{{- '<|im_start|>' + message.role + '\n' + content }}
|
| 51 |
-
{%- endif %}
|
| 52 |
{%- if message.tool_calls %}
|
| 53 |
{%- for tool_call in message.tool_calls %}
|
| 54 |
{%- if (loop.first and content) or (not loop.first) %}
|
|
@@ -83,7 +58,4 @@
|
|
| 83 |
{%- endfor %}
|
| 84 |
{%- if add_generation_prompt %}
|
| 85 |
{{- '<|im_start|>assistant\n' }}
|
| 86 |
-
{%- if enable_thinking is defined and enable_thinking is false %}
|
| 87 |
-
{{- '<think>\n\n</think>\n\n' }}
|
| 88 |
-
{%- endif %}
|
| 89 |
{%- endif %}
|
|
|
|
| 14 |
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
|
| 15 |
{%- endif %}
|
| 16 |
{%- endif %}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 17 |
{%- for message in messages %}
|
| 18 |
{%- if message.content is string %}
|
| 19 |
{%- set content = message.content %}
|
|
|
|
| 23 |
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
|
| 24 |
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
| 25 |
{%- elif message.role == "assistant" %}
|
| 26 |
+
{{- '<|im_start|>' + message.role + '\n' + content }}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 27 |
{%- if message.tool_calls %}
|
| 28 |
{%- for tool_call in message.tool_calls %}
|
| 29 |
{%- if (loop.first and content) or (not loop.first) %}
|
|
|
|
| 58 |
{%- endfor %}
|
| 59 |
{%- if add_generation_prompt %}
|
| 60 |
{{- '<|im_start|>assistant\n' }}
|
|
|
|
|
|
|
|
|
|
| 61 |
{%- endif %}
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/1090/tokenizer_config.json
CHANGED
|
@@ -231,7 +231,7 @@
|
|
| 231 |
"eos_token": "<|im_end|>",
|
| 232 |
"errors": "replace",
|
| 233 |
"extra_special_tokens": {},
|
| 234 |
-
"model_max_length":
|
| 235 |
"pad_token": "<|im_end|>",
|
| 236 |
"padding_side": "right",
|
| 237 |
"split_special_tokens": false,
|
|
|
|
| 231 |
"eos_token": "<|im_end|>",
|
| 232 |
"errors": "replace",
|
| 233 |
"extra_special_tokens": {},
|
| 234 |
+
"model_max_length": 1010000,
|
| 235 |
"pad_token": "<|im_end|>",
|
| 236 |
"padding_side": "right",
|
| 237 |
"split_special_tokens": false,
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/218/README.md
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
---
|
| 2 |
-
base_model: Qwen/Qwen3-4B
|
| 3 |
library_name: peft
|
| 4 |
pipeline_tag: text-generation
|
| 5 |
tags:
|
| 6 |
-
- base_model:adapter:Qwen/Qwen3-4B
|
| 7 |
- lora
|
| 8 |
- transformers
|
| 9 |
---
|
|
|
|
| 1 |
---
|
| 2 |
+
base_model: Qwen/Qwen3-4B-Instruct-2507
|
| 3 |
library_name: peft
|
| 4 |
pipeline_tag: text-generation
|
| 5 |
tags:
|
| 6 |
+
- base_model:adapter:Qwen/Qwen3-4B-Instruct-2507
|
| 7 |
- lora
|
| 8 |
- transformers
|
| 9 |
---
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/218/adapter_config.json
CHANGED
|
@@ -3,7 +3,7 @@
|
|
| 3 |
"alpha_pattern": {},
|
| 4 |
"arrow_config": null,
|
| 5 |
"auto_mapping": null,
|
| 6 |
-
"base_model_name_or_path": "Qwen/Qwen3-4B",
|
| 7 |
"bias": "none",
|
| 8 |
"corda_config": null,
|
| 9 |
"ensure_weight_tying": false,
|
|
@@ -29,13 +29,13 @@
|
|
| 29 |
"rank_pattern": {},
|
| 30 |
"revision": null,
|
| 31 |
"target_modules": [
|
| 32 |
-
"k_proj",
|
| 33 |
-
"q_proj",
|
| 34 |
-
"v_proj",
|
| 35 |
"down_proj",
|
| 36 |
"gate_proj",
|
| 37 |
-
"
|
| 38 |
-
"up_proj"
|
|
|
|
|
|
|
|
|
|
| 39 |
],
|
| 40 |
"target_parameters": null,
|
| 41 |
"task_type": "CAUSAL_LM",
|
|
|
|
| 3 |
"alpha_pattern": {},
|
| 4 |
"arrow_config": null,
|
| 5 |
"auto_mapping": null,
|
| 6 |
+
"base_model_name_or_path": "Qwen/Qwen3-4B-Instruct-2507",
|
| 7 |
"bias": "none",
|
| 8 |
"corda_config": null,
|
| 9 |
"ensure_weight_tying": false,
|
|
|
|
| 29 |
"rank_pattern": {},
|
| 30 |
"revision": null,
|
| 31 |
"target_modules": [
|
|
|
|
|
|
|
|
|
|
| 32 |
"down_proj",
|
| 33 |
"gate_proj",
|
| 34 |
+
"q_proj",
|
| 35 |
+
"up_proj",
|
| 36 |
+
"v_proj",
|
| 37 |
+
"k_proj",
|
| 38 |
+
"o_proj"
|
| 39 |
],
|
| 40 |
"target_parameters": null,
|
| 41 |
"task_type": "CAUSAL_LM",
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/218/adapter_model.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 132200851
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:aa57bb9340e14e59218ab048a2febcee88a4b989afdfceae602cfa90d0888705
|
| 3 |
size 132200851
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/218/chat_template.jinja
CHANGED
|
@@ -14,14 +14,6 @@
|
|
| 14 |
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
|
| 15 |
{%- endif %}
|
| 16 |
{%- endif %}
|
| 17 |
-
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
|
| 18 |
-
{%- for message in messages[::-1] %}
|
| 19 |
-
{%- set index = (messages|length - 1) - loop.index0 %}
|
| 20 |
-
{%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('<tool_response>') and message.content.endswith('</tool_response>')) %}
|
| 21 |
-
{%- set ns.multi_step_tool = false %}
|
| 22 |
-
{%- set ns.last_query_index = index %}
|
| 23 |
-
{%- endif %}
|
| 24 |
-
{%- endfor %}
|
| 25 |
{%- for message in messages %}
|
| 26 |
{%- if message.content is string %}
|
| 27 |
{%- set content = message.content %}
|
|
@@ -31,24 +23,7 @@
|
|
| 31 |
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
|
| 32 |
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
| 33 |
{%- elif message.role == "assistant" %}
|
| 34 |
-
{
|
| 35 |
-
{%- if message.reasoning_content is string %}
|
| 36 |
-
{%- set reasoning_content = message.reasoning_content %}
|
| 37 |
-
{%- else %}
|
| 38 |
-
{%- if '</think>' in content %}
|
| 39 |
-
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
|
| 40 |
-
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
|
| 41 |
-
{%- endif %}
|
| 42 |
-
{%- endif %}
|
| 43 |
-
{%- if loop.index0 > ns.last_query_index %}
|
| 44 |
-
{%- if loop.last or (not loop.last and reasoning_content) %}
|
| 45 |
-
{{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content.strip('\n') + '\n</think>\n\n' + content.lstrip('\n') }}
|
| 46 |
-
{%- else %}
|
| 47 |
-
{{- '<|im_start|>' + message.role + '\n' + content }}
|
| 48 |
-
{%- endif %}
|
| 49 |
-
{%- else %}
|
| 50 |
-
{{- '<|im_start|>' + message.role + '\n' + content }}
|
| 51 |
-
{%- endif %}
|
| 52 |
{%- if message.tool_calls %}
|
| 53 |
{%- for tool_call in message.tool_calls %}
|
| 54 |
{%- if (loop.first and content) or (not loop.first) %}
|
|
@@ -83,7 +58,4 @@
|
|
| 83 |
{%- endfor %}
|
| 84 |
{%- if add_generation_prompt %}
|
| 85 |
{{- '<|im_start|>assistant\n' }}
|
| 86 |
-
{%- if enable_thinking is defined and enable_thinking is false %}
|
| 87 |
-
{{- '<think>\n\n</think>\n\n' }}
|
| 88 |
-
{%- endif %}
|
| 89 |
{%- endif %}
|
|
|
|
| 14 |
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
|
| 15 |
{%- endif %}
|
| 16 |
{%- endif %}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 17 |
{%- for message in messages %}
|
| 18 |
{%- if message.content is string %}
|
| 19 |
{%- set content = message.content %}
|
|
|
|
| 23 |
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
|
| 24 |
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
| 25 |
{%- elif message.role == "assistant" %}
|
| 26 |
+
{{- '<|im_start|>' + message.role + '\n' + content }}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 27 |
{%- if message.tool_calls %}
|
| 28 |
{%- for tool_call in message.tool_calls %}
|
| 29 |
{%- if (loop.first and content) or (not loop.first) %}
|
|
|
|
| 58 |
{%- endfor %}
|
| 59 |
{%- if add_generation_prompt %}
|
| 60 |
{{- '<|im_start|>assistant\n' }}
|
|
|
|
|
|
|
|
|
|
| 61 |
{%- endif %}
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/218/tokenizer_config.json
CHANGED
|
@@ -231,7 +231,7 @@
|
|
| 231 |
"eos_token": "<|im_end|>",
|
| 232 |
"errors": "replace",
|
| 233 |
"extra_special_tokens": {},
|
| 234 |
-
"model_max_length":
|
| 235 |
"pad_token": "<|im_end|>",
|
| 236 |
"padding_side": "right",
|
| 237 |
"split_special_tokens": false,
|
|
|
|
| 231 |
"eos_token": "<|im_end|>",
|
| 232 |
"errors": "replace",
|
| 233 |
"extra_special_tokens": {},
|
| 234 |
+
"model_max_length": 1010000,
|
| 235 |
"pad_token": "<|im_end|>",
|
| 236 |
"padding_side": "right",
|
| 237 |
"split_special_tokens": false,
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/436/README.md
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
---
|
| 2 |
-
base_model: Qwen/Qwen3-4B
|
| 3 |
library_name: peft
|
| 4 |
pipeline_tag: text-generation
|
| 5 |
tags:
|
| 6 |
-
- base_model:adapter:Qwen/Qwen3-4B
|
| 7 |
- lora
|
| 8 |
- transformers
|
| 9 |
---
|
|
|
|
| 1 |
---
|
| 2 |
+
base_model: Qwen/Qwen3-4B-Instruct-2507
|
| 3 |
library_name: peft
|
| 4 |
pipeline_tag: text-generation
|
| 5 |
tags:
|
| 6 |
+
- base_model:adapter:Qwen/Qwen3-4B-Instruct-2507
|
| 7 |
- lora
|
| 8 |
- transformers
|
| 9 |
---
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/436/adapter_config.json
CHANGED
|
@@ -3,7 +3,7 @@
|
|
| 3 |
"alpha_pattern": {},
|
| 4 |
"arrow_config": null,
|
| 5 |
"auto_mapping": null,
|
| 6 |
-
"base_model_name_or_path": "Qwen/Qwen3-4B",
|
| 7 |
"bias": "none",
|
| 8 |
"corda_config": null,
|
| 9 |
"ensure_weight_tying": false,
|
|
@@ -29,13 +29,13 @@
|
|
| 29 |
"rank_pattern": {},
|
| 30 |
"revision": null,
|
| 31 |
"target_modules": [
|
| 32 |
-
"k_proj",
|
| 33 |
-
"q_proj",
|
| 34 |
-
"v_proj",
|
| 35 |
"down_proj",
|
| 36 |
"gate_proj",
|
| 37 |
-
"
|
| 38 |
-
"up_proj"
|
|
|
|
|
|
|
|
|
|
| 39 |
],
|
| 40 |
"target_parameters": null,
|
| 41 |
"task_type": "CAUSAL_LM",
|
|
|
|
| 3 |
"alpha_pattern": {},
|
| 4 |
"arrow_config": null,
|
| 5 |
"auto_mapping": null,
|
| 6 |
+
"base_model_name_or_path": "Qwen/Qwen3-4B-Instruct-2507",
|
| 7 |
"bias": "none",
|
| 8 |
"corda_config": null,
|
| 9 |
"ensure_weight_tying": false,
|
|
|
|
| 29 |
"rank_pattern": {},
|
| 30 |
"revision": null,
|
| 31 |
"target_modules": [
|
|
|
|
|
|
|
|
|
|
| 32 |
"down_proj",
|
| 33 |
"gate_proj",
|
| 34 |
+
"q_proj",
|
| 35 |
+
"up_proj",
|
| 36 |
+
"v_proj",
|
| 37 |
+
"k_proj",
|
| 38 |
+
"o_proj"
|
| 39 |
],
|
| 40 |
"target_parameters": null,
|
| 41 |
"task_type": "CAUSAL_LM",
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/436/adapter_model.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 132200851
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:62fa33594d3a4d460651c7bbc4edfacf2582d74c1b2247d3b14bbcb32a4d34d4
|
| 3 |
size 132200851
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/436/chat_template.jinja
CHANGED
|
@@ -14,14 +14,6 @@
|
|
| 14 |
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
|
| 15 |
{%- endif %}
|
| 16 |
{%- endif %}
|
| 17 |
-
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
|
| 18 |
-
{%- for message in messages[::-1] %}
|
| 19 |
-
{%- set index = (messages|length - 1) - loop.index0 %}
|
| 20 |
-
{%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('<tool_response>') and message.content.endswith('</tool_response>')) %}
|
| 21 |
-
{%- set ns.multi_step_tool = false %}
|
| 22 |
-
{%- set ns.last_query_index = index %}
|
| 23 |
-
{%- endif %}
|
| 24 |
-
{%- endfor %}
|
| 25 |
{%- for message in messages %}
|
| 26 |
{%- if message.content is string %}
|
| 27 |
{%- set content = message.content %}
|
|
@@ -31,24 +23,7 @@
|
|
| 31 |
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
|
| 32 |
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
| 33 |
{%- elif message.role == "assistant" %}
|
| 34 |
-
{
|
| 35 |
-
{%- if message.reasoning_content is string %}
|
| 36 |
-
{%- set reasoning_content = message.reasoning_content %}
|
| 37 |
-
{%- else %}
|
| 38 |
-
{%- if '</think>' in content %}
|
| 39 |
-
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
|
| 40 |
-
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
|
| 41 |
-
{%- endif %}
|
| 42 |
-
{%- endif %}
|
| 43 |
-
{%- if loop.index0 > ns.last_query_index %}
|
| 44 |
-
{%- if loop.last or (not loop.last and reasoning_content) %}
|
| 45 |
-
{{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content.strip('\n') + '\n</think>\n\n' + content.lstrip('\n') }}
|
| 46 |
-
{%- else %}
|
| 47 |
-
{{- '<|im_start|>' + message.role + '\n' + content }}
|
| 48 |
-
{%- endif %}
|
| 49 |
-
{%- else %}
|
| 50 |
-
{{- '<|im_start|>' + message.role + '\n' + content }}
|
| 51 |
-
{%- endif %}
|
| 52 |
{%- if message.tool_calls %}
|
| 53 |
{%- for tool_call in message.tool_calls %}
|
| 54 |
{%- if (loop.first and content) or (not loop.first) %}
|
|
@@ -83,7 +58,4 @@
|
|
| 83 |
{%- endfor %}
|
| 84 |
{%- if add_generation_prompt %}
|
| 85 |
{{- '<|im_start|>assistant\n' }}
|
| 86 |
-
{%- if enable_thinking is defined and enable_thinking is false %}
|
| 87 |
-
{{- '<think>\n\n</think>\n\n' }}
|
| 88 |
-
{%- endif %}
|
| 89 |
{%- endif %}
|
|
|
|
| 14 |
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
|
| 15 |
{%- endif %}
|
| 16 |
{%- endif %}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 17 |
{%- for message in messages %}
|
| 18 |
{%- if message.content is string %}
|
| 19 |
{%- set content = message.content %}
|
|
|
|
| 23 |
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
|
| 24 |
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
| 25 |
{%- elif message.role == "assistant" %}
|
| 26 |
+
{{- '<|im_start|>' + message.role + '\n' + content }}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 27 |
{%- if message.tool_calls %}
|
| 28 |
{%- for tool_call in message.tool_calls %}
|
| 29 |
{%- if (loop.first and content) or (not loop.first) %}
|
|
|
|
| 58 |
{%- endfor %}
|
| 59 |
{%- if add_generation_prompt %}
|
| 60 |
{{- '<|im_start|>assistant\n' }}
|
|
|
|
|
|
|
|
|
|
| 61 |
{%- endif %}
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/436/tokenizer_config.json
CHANGED
|
@@ -231,7 +231,7 @@
|
|
| 231 |
"eos_token": "<|im_end|>",
|
| 232 |
"errors": "replace",
|
| 233 |
"extra_special_tokens": {},
|
| 234 |
-
"model_max_length":
|
| 235 |
"pad_token": "<|im_end|>",
|
| 236 |
"padding_side": "right",
|
| 237 |
"split_special_tokens": false,
|
|
|
|
| 231 |
"eos_token": "<|im_end|>",
|
| 232 |
"errors": "replace",
|
| 233 |
"extra_special_tokens": {},
|
| 234 |
+
"model_max_length": 1010000,
|
| 235 |
"pad_token": "<|im_end|>",
|
| 236 |
"padding_side": "right",
|
| 237 |
"split_special_tokens": false,
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/654/README.md
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
---
|
| 2 |
-
base_model: Qwen/Qwen3-4B
|
| 3 |
library_name: peft
|
| 4 |
pipeline_tag: text-generation
|
| 5 |
tags:
|
| 6 |
-
- base_model:adapter:Qwen/Qwen3-4B
|
| 7 |
- lora
|
| 8 |
- transformers
|
| 9 |
---
|
|
|
|
| 1 |
---
|
| 2 |
+
base_model: Qwen/Qwen3-4B-Instruct-2507
|
| 3 |
library_name: peft
|
| 4 |
pipeline_tag: text-generation
|
| 5 |
tags:
|
| 6 |
+
- base_model:adapter:Qwen/Qwen3-4B-Instruct-2507
|
| 7 |
- lora
|
| 8 |
- transformers
|
| 9 |
---
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/654/adapter_config.json
CHANGED
|
@@ -3,7 +3,7 @@
|
|
| 3 |
"alpha_pattern": {},
|
| 4 |
"arrow_config": null,
|
| 5 |
"auto_mapping": null,
|
| 6 |
-
"base_model_name_or_path": "Qwen/Qwen3-4B",
|
| 7 |
"bias": "none",
|
| 8 |
"corda_config": null,
|
| 9 |
"ensure_weight_tying": false,
|
|
@@ -29,13 +29,13 @@
|
|
| 29 |
"rank_pattern": {},
|
| 30 |
"revision": null,
|
| 31 |
"target_modules": [
|
| 32 |
-
"k_proj",
|
| 33 |
-
"q_proj",
|
| 34 |
-
"v_proj",
|
| 35 |
"down_proj",
|
| 36 |
"gate_proj",
|
| 37 |
-
"
|
| 38 |
-
"up_proj"
|
|
|
|
|
|
|
|
|
|
| 39 |
],
|
| 40 |
"target_parameters": null,
|
| 41 |
"task_type": "CAUSAL_LM",
|
|
|
|
| 3 |
"alpha_pattern": {},
|
| 4 |
"arrow_config": null,
|
| 5 |
"auto_mapping": null,
|
| 6 |
+
"base_model_name_or_path": "Qwen/Qwen3-4B-Instruct-2507",
|
| 7 |
"bias": "none",
|
| 8 |
"corda_config": null,
|
| 9 |
"ensure_weight_tying": false,
|
|
|
|
| 29 |
"rank_pattern": {},
|
| 30 |
"revision": null,
|
| 31 |
"target_modules": [
|
|
|
|
|
|
|
|
|
|
| 32 |
"down_proj",
|
| 33 |
"gate_proj",
|
| 34 |
+
"q_proj",
|
| 35 |
+
"up_proj",
|
| 36 |
+
"v_proj",
|
| 37 |
+
"k_proj",
|
| 38 |
+
"o_proj"
|
| 39 |
],
|
| 40 |
"target_parameters": null,
|
| 41 |
"task_type": "CAUSAL_LM",
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/654/adapter_model.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 132200851
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:95cfd0f09243e85433b6fb67eeb39699df8dfacb808c119b24042cd0f3a2c6bf
|
| 3 |
size 132200851
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/654/chat_template.jinja
CHANGED
|
@@ -14,14 +14,6 @@
|
|
| 14 |
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
|
| 15 |
{%- endif %}
|
| 16 |
{%- endif %}
|
| 17 |
-
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
|
| 18 |
-
{%- for message in messages[::-1] %}
|
| 19 |
-
{%- set index = (messages|length - 1) - loop.index0 %}
|
| 20 |
-
{%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('<tool_response>') and message.content.endswith('</tool_response>')) %}
|
| 21 |
-
{%- set ns.multi_step_tool = false %}
|
| 22 |
-
{%- set ns.last_query_index = index %}
|
| 23 |
-
{%- endif %}
|
| 24 |
-
{%- endfor %}
|
| 25 |
{%- for message in messages %}
|
| 26 |
{%- if message.content is string %}
|
| 27 |
{%- set content = message.content %}
|
|
@@ -31,24 +23,7 @@
|
|
| 31 |
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
|
| 32 |
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
| 33 |
{%- elif message.role == "assistant" %}
|
| 34 |
-
{
|
| 35 |
-
{%- if message.reasoning_content is string %}
|
| 36 |
-
{%- set reasoning_content = message.reasoning_content %}
|
| 37 |
-
{%- else %}
|
| 38 |
-
{%- if '</think>' in content %}
|
| 39 |
-
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
|
| 40 |
-
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
|
| 41 |
-
{%- endif %}
|
| 42 |
-
{%- endif %}
|
| 43 |
-
{%- if loop.index0 > ns.last_query_index %}
|
| 44 |
-
{%- if loop.last or (not loop.last and reasoning_content) %}
|
| 45 |
-
{{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content.strip('\n') + '\n</think>\n\n' + content.lstrip('\n') }}
|
| 46 |
-
{%- else %}
|
| 47 |
-
{{- '<|im_start|>' + message.role + '\n' + content }}
|
| 48 |
-
{%- endif %}
|
| 49 |
-
{%- else %}
|
| 50 |
-
{{- '<|im_start|>' + message.role + '\n' + content }}
|
| 51 |
-
{%- endif %}
|
| 52 |
{%- if message.tool_calls %}
|
| 53 |
{%- for tool_call in message.tool_calls %}
|
| 54 |
{%- if (loop.first and content) or (not loop.first) %}
|
|
@@ -83,7 +58,4 @@
|
|
| 83 |
{%- endfor %}
|
| 84 |
{%- if add_generation_prompt %}
|
| 85 |
{{- '<|im_start|>assistant\n' }}
|
| 86 |
-
{%- if enable_thinking is defined and enable_thinking is false %}
|
| 87 |
-
{{- '<think>\n\n</think>\n\n' }}
|
| 88 |
-
{%- endif %}
|
| 89 |
{%- endif %}
|
|
|
|
| 14 |
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
|
| 15 |
{%- endif %}
|
| 16 |
{%- endif %}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 17 |
{%- for message in messages %}
|
| 18 |
{%- if message.content is string %}
|
| 19 |
{%- set content = message.content %}
|
|
|
|
| 23 |
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
|
| 24 |
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
| 25 |
{%- elif message.role == "assistant" %}
|
| 26 |
+
{{- '<|im_start|>' + message.role + '\n' + content }}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 27 |
{%- if message.tool_calls %}
|
| 28 |
{%- for tool_call in message.tool_calls %}
|
| 29 |
{%- if (loop.first and content) or (not loop.first) %}
|
|
|
|
| 58 |
{%- endfor %}
|
| 59 |
{%- if add_generation_prompt %}
|
| 60 |
{{- '<|im_start|>assistant\n' }}
|
|
|
|
|
|
|
|
|
|
| 61 |
{%- endif %}
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/654/tokenizer_config.json
CHANGED
|
@@ -231,7 +231,7 @@
|
|
| 231 |
"eos_token": "<|im_end|>",
|
| 232 |
"errors": "replace",
|
| 233 |
"extra_special_tokens": {},
|
| 234 |
-
"model_max_length":
|
| 235 |
"pad_token": "<|im_end|>",
|
| 236 |
"padding_side": "right",
|
| 237 |
"split_special_tokens": false,
|
|
|
|
| 231 |
"eos_token": "<|im_end|>",
|
| 232 |
"errors": "replace",
|
| 233 |
"extra_special_tokens": {},
|
| 234 |
+
"model_max_length": 1010000,
|
| 235 |
"pad_token": "<|im_end|>",
|
| 236 |
"padding_side": "right",
|
| 237 |
"split_special_tokens": false,
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/872/README.md
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
---
|
| 2 |
-
base_model: Qwen/Qwen3-4B
|
| 3 |
library_name: peft
|
| 4 |
pipeline_tag: text-generation
|
| 5 |
tags:
|
| 6 |
-
- base_model:adapter:Qwen/Qwen3-4B
|
| 7 |
- lora
|
| 8 |
- transformers
|
| 9 |
---
|
|
|
|
| 1 |
---
|
| 2 |
+
base_model: Qwen/Qwen3-4B-Instruct-2507
|
| 3 |
library_name: peft
|
| 4 |
pipeline_tag: text-generation
|
| 5 |
tags:
|
| 6 |
+
- base_model:adapter:Qwen/Qwen3-4B-Instruct-2507
|
| 7 |
- lora
|
| 8 |
- transformers
|
| 9 |
---
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/872/adapter_config.json
CHANGED
|
@@ -3,7 +3,7 @@
|
|
| 3 |
"alpha_pattern": {},
|
| 4 |
"arrow_config": null,
|
| 5 |
"auto_mapping": null,
|
| 6 |
-
"base_model_name_or_path": "Qwen/Qwen3-4B",
|
| 7 |
"bias": "none",
|
| 8 |
"corda_config": null,
|
| 9 |
"ensure_weight_tying": false,
|
|
@@ -29,13 +29,13 @@
|
|
| 29 |
"rank_pattern": {},
|
| 30 |
"revision": null,
|
| 31 |
"target_modules": [
|
| 32 |
-
"k_proj",
|
| 33 |
-
"q_proj",
|
| 34 |
-
"v_proj",
|
| 35 |
"down_proj",
|
| 36 |
"gate_proj",
|
| 37 |
-
"
|
| 38 |
-
"up_proj"
|
|
|
|
|
|
|
|
|
|
| 39 |
],
|
| 40 |
"target_parameters": null,
|
| 41 |
"task_type": "CAUSAL_LM",
|
|
|
|
| 3 |
"alpha_pattern": {},
|
| 4 |
"arrow_config": null,
|
| 5 |
"auto_mapping": null,
|
| 6 |
+
"base_model_name_or_path": "Qwen/Qwen3-4B-Instruct-2507",
|
| 7 |
"bias": "none",
|
| 8 |
"corda_config": null,
|
| 9 |
"ensure_weight_tying": false,
|
|
|
|
| 29 |
"rank_pattern": {},
|
| 30 |
"revision": null,
|
| 31 |
"target_modules": [
|
|
|
|
|
|
|
|
|
|
| 32 |
"down_proj",
|
| 33 |
"gate_proj",
|
| 34 |
+
"q_proj",
|
| 35 |
+
"up_proj",
|
| 36 |
+
"v_proj",
|
| 37 |
+
"k_proj",
|
| 38 |
+
"o_proj"
|
| 39 |
],
|
| 40 |
"target_parameters": null,
|
| 41 |
"task_type": "CAUSAL_LM",
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/872/adapter_model.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 132200851
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:24f1f070ca1f20cea2b2d4b9d0ab1ebcc42dc23f6e84e42700dea2fc20828a69
|
| 3 |
size 132200851
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/872/chat_template.jinja
CHANGED
|
@@ -14,14 +14,6 @@
|
|
| 14 |
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
|
| 15 |
{%- endif %}
|
| 16 |
{%- endif %}
|
| 17 |
-
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
|
| 18 |
-
{%- for message in messages[::-1] %}
|
| 19 |
-
{%- set index = (messages|length - 1) - loop.index0 %}
|
| 20 |
-
{%- if ns.multi_step_tool and message.role == "user" and message.content is string and not(message.content.startswith('<tool_response>') and message.content.endswith('</tool_response>')) %}
|
| 21 |
-
{%- set ns.multi_step_tool = false %}
|
| 22 |
-
{%- set ns.last_query_index = index %}
|
| 23 |
-
{%- endif %}
|
| 24 |
-
{%- endfor %}
|
| 25 |
{%- for message in messages %}
|
| 26 |
{%- if message.content is string %}
|
| 27 |
{%- set content = message.content %}
|
|
@@ -31,24 +23,7 @@
|
|
| 31 |
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
|
| 32 |
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
| 33 |
{%- elif message.role == "assistant" %}
|
| 34 |
-
{
|
| 35 |
-
{%- if message.reasoning_content is string %}
|
| 36 |
-
{%- set reasoning_content = message.reasoning_content %}
|
| 37 |
-
{%- else %}
|
| 38 |
-
{%- if '</think>' in content %}
|
| 39 |
-
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
|
| 40 |
-
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
|
| 41 |
-
{%- endif %}
|
| 42 |
-
{%- endif %}
|
| 43 |
-
{%- if loop.index0 > ns.last_query_index %}
|
| 44 |
-
{%- if loop.last or (not loop.last and reasoning_content) %}
|
| 45 |
-
{{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content.strip('\n') + '\n</think>\n\n' + content.lstrip('\n') }}
|
| 46 |
-
{%- else %}
|
| 47 |
-
{{- '<|im_start|>' + message.role + '\n' + content }}
|
| 48 |
-
{%- endif %}
|
| 49 |
-
{%- else %}
|
| 50 |
-
{{- '<|im_start|>' + message.role + '\n' + content }}
|
| 51 |
-
{%- endif %}
|
| 52 |
{%- if message.tool_calls %}
|
| 53 |
{%- for tool_call in message.tool_calls %}
|
| 54 |
{%- if (loop.first and content) or (not loop.first) %}
|
|
@@ -83,7 +58,4 @@
|
|
| 83 |
{%- endfor %}
|
| 84 |
{%- if add_generation_prompt %}
|
| 85 |
{{- '<|im_start|>assistant\n' }}
|
| 86 |
-
{%- if enable_thinking is defined and enable_thinking is false %}
|
| 87 |
-
{{- '<think>\n\n</think>\n\n' }}
|
| 88 |
-
{%- endif %}
|
| 89 |
{%- endif %}
|
|
|
|
| 14 |
{{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
|
| 15 |
{%- endif %}
|
| 16 |
{%- endif %}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 17 |
{%- for message in messages %}
|
| 18 |
{%- if message.content is string %}
|
| 19 |
{%- set content = message.content %}
|
|
|
|
| 23 |
{%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
|
| 24 |
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
| 25 |
{%- elif message.role == "assistant" %}
|
| 26 |
+
{{- '<|im_start|>' + message.role + '\n' + content }}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 27 |
{%- if message.tool_calls %}
|
| 28 |
{%- for tool_call in message.tool_calls %}
|
| 29 |
{%- if (loop.first and content) or (not loop.first) %}
|
|
|
|
| 58 |
{%- endfor %}
|
| 59 |
{%- if add_generation_prompt %}
|
| 60 |
{{- '<|im_start|>assistant\n' }}
|
|
|
|
|
|
|
|
|
|
| 61 |
{%- endif %}
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/872/tokenizer_config.json
CHANGED
|
@@ -231,7 +231,7 @@
|
|
| 231 |
"eos_token": "<|im_end|>",
|
| 232 |
"errors": "replace",
|
| 233 |
"extra_special_tokens": {},
|
| 234 |
-
"model_max_length":
|
| 235 |
"pad_token": "<|im_end|>",
|
| 236 |
"padding_side": "right",
|
| 237 |
"split_special_tokens": false,
|
|
|
|
| 231 |
"eos_token": "<|im_end|>",
|
| 232 |
"errors": "replace",
|
| 233 |
"extra_special_tokens": {},
|
| 234 |
+
"model_max_length": 1010000,
|
| 235 |
"pad_token": "<|im_end|>",
|
| 236 |
"padding_side": "right",
|
| 237 |
"split_special_tokens": false,
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/args.json
CHANGED
|
@@ -1 +1 @@
|
|
| 1 |
-
{"model_path": "Qwen/Qwen3-4B", "ckpt_name": "qwen3-4B", "model_type": "qwen", "teacher_model_type": null, "n_gpu": 2, "n_nodes": 1, "teacher_model_path": null, "teacher_ckpt_name": null, "teacher_model_fp16": false, "model_parallel": false, "model_parallel_size": null, "no_value": false, "dropout_path_rate": null, "fp32": false, "bf16": false, "type": "lm", "do_train": true, "do_valid": true, "do_eval": false, "base_path": ".", "load": null, "save": "./results/qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1", "log_interval": 20, "mid_log_num": -1, "save_interval": -1, "eval_interval": -1, "local_rank": 0, "save_additional_suffix": "", "save_rollout": false, "eb_sample_times": 3, "split": null, "data_dir": "processed_data/benchmarks/spider_data/qwen", "processed_data_dir": null, "force_process": false, "force_process_demo": false, "data_process_workers": -1, "train_num": -1, "train_ratio": 1, "dev_num": -1, "dev_ratio": 1, "gen_num": -1, "data_names": null, "prompt_type": null, "num_workers": 0, "max_prompt_length": 1479, "t_max_prompt_length": 640, "min_prompt_length": 128, "json_data": false, "bin_data": false, "txt_data": false, "prompt_data_dir": null, "lm_data_dir": null, "eval_ppl": false, "eval_rw": false, "eval_gen": true, "only_prompt": false, "slice_data": false, "batch_size": 4, "eval_batch_size": 16, "clip_grad": 1.0, "total_iters": null, "train_iters_per_epoch": -1, "max_length": 1612, "t_max_length": 1024, "seed": 42, "seed_order": 42, "seed_data": 42, "seed_ppo": 42, "seed_lm": 7, "epochs": 5, "training_epochs": 10000, "gradient_accumulation_steps": 4, "gradient_checkpointing": true, "attn_dtype": null, "lr": 0.0001, "lr_min": 1e-07, "weight_decay": 0.01, "loss_scale": 65536, "kd_ratio": null, "warmup_iters": 0, "warmup_ratio": 0.1, "lr_decay_iters": null, "lr_decay_style": "wrmup_cosine", "scheduler_name": "constant_trm", "w_span_loss": 1.0, "reward_scaling": null, "cliprange_reward": 1, "ppo_epochs": null, "num_rollouts": 256, "num_rollouts_per_device": null, "cliprange": 0.2, "chunk_size": null, "gamma": 0.95, "length_norm": false, "single_step_reg": false, "teacher_mixed_alpha": null, "lm_coef": 1, "skew_alpha": 0.1, "student_gen": false, "gen_top_p": 1.0, "gen_num_beams": 2, "mixed_alpha": 0.5, "loss_eps": 0.1, "init_threshold": 0.0, "capacity": 1000, "replay_ratio": "decreasing", "student_layer_mapping": [-1], "teacher_layer_mapping": [-1], "split_layer_mapping": [0, 0, 0, 0], "use_dsa": false, "use_hs": false, "top_k": 0, "top_p": 0.95, "do_sample": true, "no_repeat_ngram_size": 6, "repetition_penalty": null, "num_beams": 1, "temperature": 0.5, "peft": "lora", "peft_lora_r": 32, "peft_lora_alpha": 64, "peft_lora_dropout": 0.1, "peft_name": null, "peft_path": null, "teacher_peft_name": null, "teacher_peft_path": null, "deepspeed": true, "deepspeed_config": "./configs/deepspeed/ds_config_bf16.json", "deepscale": false, "deepscale_config": null, "rank": 0, "world_size": 2}
|
|
|
|
| 1 |
+
{"model_path": "Qwen/Qwen3-4B-Instruct-2507", "ckpt_name": "qwen3-4B", "model_type": "qwen", "teacher_model_type": null, "n_gpu": 2, "n_nodes": 1, "teacher_model_path": null, "teacher_ckpt_name": null, "teacher_model_fp16": false, "model_parallel": false, "model_parallel_size": null, "no_value": false, "dropout_path_rate": null, "fp32": false, "bf16": false, "type": "lm", "do_train": true, "do_valid": true, "do_eval": false, "base_path": ".", "load": null, "save": "./results/qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1", "log_interval": 20, "mid_log_num": -1, "save_interval": -1, "eval_interval": -1, "local_rank": 0, "save_additional_suffix": "", "save_rollout": false, "eb_sample_times": 3, "split": null, "data_dir": "processed_data/benchmarks/spider_data/qwen", "processed_data_dir": null, "force_process": false, "force_process_demo": false, "data_process_workers": -1, "train_num": -1, "train_ratio": 1, "dev_num": -1, "dev_ratio": 1, "gen_num": -1, "data_names": null, "prompt_type": null, "num_workers": 0, "max_prompt_length": 1479, "t_max_prompt_length": 640, "min_prompt_length": 128, "json_data": false, "bin_data": false, "txt_data": false, "prompt_data_dir": null, "lm_data_dir": null, "eval_ppl": false, "eval_rw": false, "eval_gen": true, "only_prompt": false, "slice_data": false, "batch_size": 4, "eval_batch_size": 16, "clip_grad": 1.0, "total_iters": null, "train_iters_per_epoch": -1, "max_length": 1612, "t_max_length": 1024, "seed": 42, "seed_order": 42, "seed_data": 42, "seed_ppo": 42, "seed_lm": 7, "epochs": 5, "training_epochs": 10000, "gradient_accumulation_steps": 4, "gradient_checkpointing": true, "attn_dtype": null, "lr": 0.0001, "lr_min": 1e-07, "weight_decay": 0.01, "loss_scale": 65536, "kd_ratio": null, "warmup_iters": 0, "warmup_ratio": 0.1, "lr_decay_iters": null, "lr_decay_style": "wrmup_cosine", "scheduler_name": "constant_trm", "w_span_loss": 1.0, "reward_scaling": null, "cliprange_reward": 1, "ppo_epochs": null, "num_rollouts": 256, "num_rollouts_per_device": null, "cliprange": 0.2, "chunk_size": null, "gamma": 0.95, "length_norm": false, "single_step_reg": false, "teacher_mixed_alpha": null, "lm_coef": 1, "skew_alpha": 0.1, "student_gen": false, "gen_top_p": 1.0, "gen_num_beams": 2, "mixed_alpha": 0.5, "loss_eps": 0.1, "init_threshold": 0.0, "capacity": 1000, "replay_ratio": "decreasing", "student_layer_mapping": [-1], "teacher_layer_mapping": [-1], "split_layer_mapping": [0, 0, 0, 0], "use_dsa": false, "use_hs": false, "top_k": 0, "top_p": 0.95, "do_sample": true, "no_repeat_ngram_size": 6, "repetition_penalty": null, "num_beams": 1, "temperature": 0.5, "peft": "lora", "peft_lora_r": 32, "peft_lora_alpha": 64, "peft_lora_dropout": 0.1, "peft_name": null, "peft_path": null, "teacher_peft_name": null, "teacher_peft_path": null, "deepspeed": true, "deepspeed_config": "./configs/deepspeed/ds_config_bf16.json", "deepscale": false, "deepscale_config": null, "rank": 0, "world_size": 2}
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/eval/0/answers.jsonl
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/eval/1/answers.jsonl
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/eval/2/answers.jsonl
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/eval/3/answers.jsonl
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/eval/4/answers.jsonl
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
qwen3/sft_sft_qwen3_4b_spider_lora/e5-bs4-lr0.0001-G4-N2-NN1-lora-32-64-0.1/log.txt
CHANGED
|
@@ -61,3 +61,66 @@ train | epoch 4 | Iter: 4156/ 4360 | global iter: 1040/ 1090 | loss: 0.0
|
|
| 61 |
train | epoch 4 | Iter: 4236/ 4360 | global iter: 1060/ 1090 | loss: 0.0028 | ds_loss: 0.0000 | lr: 2.4619e-07 | scale: 1.0000 | micro time: 5.292 | step time: 21.087
|
| 62 |
train | epoch 4 | Iter: 4316/ 4360 | global iter: 1080/ 1090 | loss: 0.0036 | ds_loss: 0.0000 | lr: 3.1020e-08 | scale: 1.0000 | micro time: 5.287 | step time: 21.087
|
| 63 |
dev | avg_loss: 0.4173325047348485 | {'exact_match': 52.9981, 'rougeL': 88.8055}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 61 |
train | epoch 4 | Iter: 4236/ 4360 | global iter: 1060/ 1090 | loss: 0.0028 | ds_loss: 0.0000 | lr: 2.4619e-07 | scale: 1.0000 | micro time: 5.292 | step time: 21.087
|
| 62 |
train | epoch 4 | Iter: 4316/ 4360 | global iter: 1080/ 1090 | loss: 0.0036 | ds_loss: 0.0000 | lr: 3.1020e-08 | scale: 1.0000 | micro time: 5.287 | step time: 21.087
|
| 63 |
dev | avg_loss: 0.4173325047348485 | {'exact_match': 52.9981, 'rougeL': 88.8055}
|
| 64 |
+
|
| 65 |
+
|
| 66 |
+
============================== EXP at 2026-06-12 05:29:11 ==============================
|
| 67 |
+
dev | avg_loss: 2.152462121212121 | {'exact_match': 21.0832, 'rougeL': 71.8971}
|
| 68 |
+
train | epoch 0 | Iter: 76/ 4360 | global iter: 20/ 1090 | loss: 1.6727 | ds_loss: 0.0000 | lr: 1.7431e-05 | scale: 1.0000 | micro time: 5.289 | step time: 20.024
|
| 69 |
+
train | epoch 0 | Iter: 156/ 4360 | global iter: 40/ 1090 | loss: 0.4101 | ds_loss: 0.0000 | lr: 3.5780e-05 | scale: 1.0000 | micro time: 5.287 | step time: 21.078
|
| 70 |
+
train | epoch 0 | Iter: 236/ 4360 | global iter: 60/ 1090 | loss: 0.2133 | ds_loss: 0.0000 | lr: 5.4128e-05 | scale: 1.0000 | micro time: 5.288 | step time: 21.079
|
| 71 |
+
train | epoch 0 | Iter: 316/ 4360 | global iter: 80/ 1090 | loss: 0.1717 | ds_loss: 0.0000 | lr: 7.2477e-05 | scale: 1.0000 | micro time: 5.281 | step time: 21.078
|
| 72 |
+
train | epoch 0 | Iter: 396/ 4360 | global iter: 100/ 1090 | loss: 0.1598 | ds_loss: 0.0000 | lr: 9.0826e-05 | scale: 1.0000 | micro time: 5.286 | step time: 21.081
|
| 73 |
+
train | epoch 0 | Iter: 476/ 4360 | global iter: 120/ 1090 | loss: 0.1446 | ds_loss: 0.0000 | lr: 9.9974e-05 | scale: 1.0000 | micro time: 5.283 | step time: 21.080
|
| 74 |
+
train | epoch 0 | Iter: 556/ 4360 | global iter: 140/ 1090 | loss: 0.1296 | ds_loss: 0.0000 | lr: 9.9769e-05 | scale: 1.0000 | micro time: 5.284 | step time: 21.081
|
| 75 |
+
train | epoch 0 | Iter: 636/ 4360 | global iter: 160/ 1090 | loss: 0.1280 | ds_loss: 0.0000 | lr: 9.9360e-05 | scale: 1.0000 | micro time: 5.284 | step time: 21.076
|
| 76 |
+
train | epoch 0 | Iter: 716/ 4360 | global iter: 180/ 1090 | loss: 0.1159 | ds_loss: 0.0000 | lr: 9.8749e-05 | scale: 1.0000 | micro time: 5.285 | step time: 21.081
|
| 77 |
+
train | epoch 0 | Iter: 796/ 4360 | global iter: 200/ 1090 | loss: 0.1104 | ds_loss: 0.0000 | lr: 9.7938e-05 | scale: 1.0000 | micro time: 5.283 | step time: 21.077
|
| 78 |
+
dev | avg_loss: 0.19998816287878787 | {'exact_match': 53.7718, 'rougeL': 88.4715}
|
| 79 |
+
train | epoch 1 | Iter: 876/ 4360 | global iter: 220/ 1090 | loss: 0.1118 | ds_loss: 0.0000 | lr: 9.6930e-05 | scale: 1.0000 | micro time: 5.281 | step time: 21.079
|
| 80 |
+
train | epoch 1 | Iter: 956/ 4360 | global iter: 240/ 1090 | loss: 0.0764 | ds_loss: 0.0000 | lr: 9.5729e-05 | scale: 1.0000 | micro time: 5.282 | step time: 21.082
|
| 81 |
+
train | epoch 1 | Iter: 1036/ 4360 | global iter: 260/ 1090 | loss: 0.0793 | ds_loss: 0.0000 | lr: 9.4341e-05 | scale: 1.0000 | micro time: 5.288 | step time: 21.082
|
| 82 |
+
train | epoch 1 | Iter: 1116/ 4360 | global iter: 280/ 1090 | loss: 0.0756 | ds_loss: 0.0000 | lr: 9.2772e-05 | scale: 1.0000 | micro time: 5.290 | step time: 21.085
|
| 83 |
+
train | epoch 1 | Iter: 1196/ 4360 | global iter: 300/ 1090 | loss: 0.0735 | ds_loss: 0.0000 | lr: 9.1026e-05 | scale: 1.0000 | micro time: 5.285 | step time: 21.085
|
| 84 |
+
train | epoch 1 | Iter: 1276/ 4360 | global iter: 320/ 1090 | loss: 0.0662 | ds_loss: 0.0000 | lr: 8.9113e-05 | scale: 1.0000 | micro time: 5.287 | step time: 21.080
|
| 85 |
+
train | epoch 1 | Iter: 1356/ 4360 | global iter: 340/ 1090 | loss: 0.0659 | ds_loss: 0.0000 | lr: 8.7039e-05 | scale: 1.0000 | micro time: 5.291 | step time: 21.081
|
| 86 |
+
train | epoch 1 | Iter: 1436/ 4360 | global iter: 360/ 1090 | loss: 0.0668 | ds_loss: 0.0000 | lr: 8.4813e-05 | scale: 1.0000 | micro time: 5.290 | step time: 21.085
|
| 87 |
+
train | epoch 1 | Iter: 1516/ 4360 | global iter: 380/ 1090 | loss: 0.0683 | ds_loss: 0.0000 | lr: 8.2445e-05 | scale: 1.0000 | micro time: 5.290 | step time: 21.083
|
| 88 |
+
train | epoch 1 | Iter: 1596/ 4360 | global iter: 400/ 1090 | loss: 0.0688 | ds_loss: 0.0000 | lr: 7.9943e-05 | scale: 1.0000 | micro time: 5.288 | step time: 21.082
|
| 89 |
+
train | epoch 1 | Iter: 1676/ 4360 | global iter: 420/ 1090 | loss: 0.0608 | ds_loss: 0.0000 | lr: 7.7319e-05 | scale: 1.0000 | micro time: 5.288 | step time: 21.085
|
| 90 |
+
dev | avg_loss: 0.23133433948863635 | {'exact_match': 52.7079, 'rougeL': 88.6921}
|
| 91 |
+
train | epoch 2 | Iter: 1756/ 4360 | global iter: 440/ 1090 | loss: 0.0584 | ds_loss: 0.0000 | lr: 7.4583e-05 | scale: 1.0000 | micro time: 5.288 | step time: 21.081
|
| 92 |
+
train | epoch 2 | Iter: 1836/ 4360 | global iter: 460/ 1090 | loss: 0.0332 | ds_loss: 0.0000 | lr: 7.1746e-05 | scale: 1.0000 | micro time: 5.283 | step time: 21.079
|
| 93 |
+
train | epoch 2 | Iter: 1916/ 4360 | global iter: 480/ 1090 | loss: 0.0367 | ds_loss: 0.0000 | lr: 6.8819e-05 | scale: 1.0000 | micro time: 5.282 | step time: 21.081
|
| 94 |
+
train | epoch 2 | Iter: 1996/ 4360 | global iter: 500/ 1090 | loss: 0.0354 | ds_loss: 0.0000 | lr: 6.5816e-05 | scale: 1.0000 | micro time: 5.285 | step time: 21.081
|
| 95 |
+
train | epoch 2 | Iter: 2076/ 4360 | global iter: 520/ 1090 | loss: 0.0334 | ds_loss: 0.0000 | lr: 6.2748e-05 | scale: 1.0000 | micro time: 5.287 | step time: 21.078
|
| 96 |
+
train | epoch 2 | Iter: 2156/ 4360 | global iter: 540/ 1090 | loss: 0.0299 | ds_loss: 0.0000 | lr: 5.9627e-05 | scale: 1.0000 | micro time: 5.289 | step time: 21.079
|
| 97 |
+
train | epoch 2 | Iter: 2236/ 4360 | global iter: 560/ 1090 | loss: 0.0307 | ds_loss: 0.0000 | lr: 5.6467e-05 | scale: 1.0000 | micro time: 5.285 | step time: 21.084
|
| 98 |
+
train | epoch 2 | Iter: 2316/ 4360 | global iter: 580/ 1090 | loss: 0.0303 | ds_loss: 0.0000 | lr: 5.3280e-05 | scale: 1.0000 | micro time: 5.281 | step time: 21.079
|
| 99 |
+
train | epoch 2 | Iter: 2396/ 4360 | global iter: 600/ 1090 | loss: 0.0244 | ds_loss: 0.0000 | lr: 5.0080e-05 | scale: 1.0000 | micro time: 5.281 | step time: 21.080
|
| 100 |
+
train | epoch 2 | Iter: 2476/ 4360 | global iter: 620/ 1090 | loss: 0.0311 | ds_loss: 0.0000 | lr: 4.6880e-05 | scale: 1.0000 | micro time: 5.285 | step time: 21.081
|
| 101 |
+
train | epoch 2 | Iter: 2556/ 4360 | global iter: 640/ 1090 | loss: 0.0267 | ds_loss: 0.0000 | lr: 4.3692e-05 | scale: 1.0000 | micro time: 5.283 | step time: 21.079
|
| 102 |
+
dev | avg_loss: 0.2874348958333333 | {'exact_match': 52.0309, 'rougeL': 88.7671}
|
| 103 |
+
train | epoch 3 | Iter: 2636/ 4360 | global iter: 660/ 1090 | loss: 0.0236 | ds_loss: 0.0000 | lr: 4.0530e-05 | scale: 1.0000 | micro time: 5.284 | step time: 21.080
|
| 104 |
+
train | epoch 3 | Iter: 2716/ 4360 | global iter: 680/ 1090 | loss: 0.0128 | ds_loss: 0.0000 | lr: 3.7407e-05 | scale: 1.0000 | micro time: 5.289 | step time: 21.086
|
| 105 |
+
train | epoch 3 | Iter: 2796/ 4360 | global iter: 700/ 1090 | loss: 0.0114 | ds_loss: 0.0000 | lr: 3.4336e-05 | scale: 1.0000 | micro time: 5.288 | step time: 21.082
|
| 106 |
+
train | epoch 3 | Iter: 2876/ 4360 | global iter: 720/ 1090 | loss: 0.0114 | ds_loss: 0.0000 | lr: 3.1329e-05 | scale: 1.0000 | micro time: 5.287 | step time: 21.085
|
| 107 |
+
train | epoch 3 | Iter: 2956/ 4360 | global iter: 740/ 1090 | loss: 0.0101 | ds_loss: 0.0000 | lr: 2.8399e-05 | scale: 1.0000 | micro time: 5.285 | step time: 21.082
|
| 108 |
+
train | epoch 3 | Iter: 3036/ 4360 | global iter: 760/ 1090 | loss: 0.0110 | ds_loss: 0.0000 | lr: 2.5557e-05 | scale: 1.0000 | micro time: 5.285 | step time: 21.084
|
| 109 |
+
train | epoch 3 | Iter: 3116/ 4360 | global iter: 780/ 1090 | loss: 0.0100 | ds_loss: 0.0000 | lr: 2.2815e-05 | scale: 1.0000 | micro time: 5.283 | step time: 21.084
|
| 110 |
+
train | epoch 3 | Iter: 3196/ 4360 | global iter: 800/ 1090 | loss: 0.0103 | ds_loss: 0.0000 | lr: 2.0185e-05 | scale: 1.0000 | micro time: 5.290 | step time: 21.084
|
| 111 |
+
train | epoch 3 | Iter: 3276/ 4360 | global iter: 820/ 1090 | loss: 0.0093 | ds_loss: 0.0000 | lr: 1.7677e-05 | scale: 1.0000 | micro time: 5.291 | step time: 21.085
|
| 112 |
+
train | epoch 3 | Iter: 3356/ 4360 | global iter: 840/ 1090 | loss: 0.0087 | ds_loss: 0.0000 | lr: 1.5302e-05 | scale: 1.0000 | micro time: 5.285 | step time: 21.085
|
| 113 |
+
train | epoch 3 | Iter: 3436/ 4360 | global iter: 860/ 1090 | loss: 0.0096 | ds_loss: 0.0000 | lr: 1.3069e-05 | scale: 1.0000 | micro time: 5.287 | step time: 21.087
|
| 114 |
+
dev | avg_loss: 0.3450520833333333 | {'exact_match': 52.6112, 'rougeL': 88.9668}
|
| 115 |
+
train | epoch 4 | Iter: 3516/ 4360 | global iter: 880/ 1090 | loss: 0.0077 | ds_loss: 0.0000 | lr: 1.0987e-05 | scale: 1.0000 | micro time: 5.285 | step time: 21.081
|
| 116 |
+
train | epoch 4 | Iter: 3596/ 4360 | global iter: 900/ 1090 | loss: 0.0041 | ds_loss: 0.0000 | lr: 9.0654e-06 | scale: 1.0000 | micro time: 5.286 | step time: 21.084
|
| 117 |
+
train | epoch 4 | Iter: 3676/ 4360 | global iter: 920/ 1090 | loss: 0.0029 | ds_loss: 0.0000 | lr: 7.3116e-06 | scale: 1.0000 | micro time: 5.287 | step time: 21.084
|
| 118 |
+
train | epoch 4 | Iter: 3756/ 4360 | global iter: 940/ 1090 | loss: 0.0039 | ds_loss: 0.0000 | lr: 5.7329e-06 | scale: 1.0000 | micro time: 5.285 | step time: 21.084
|
| 119 |
+
train | epoch 4 | Iter: 3836/ 4360 | global iter: 960/ 1090 | loss: 0.0035 | ds_loss: 0.0000 | lr: 4.3358e-06 | scale: 1.0000 | micro time: 5.282 | step time: 21.085
|
| 120 |
+
train | epoch 4 | Iter: 3916/ 4360 | global iter: 980/ 1090 | loss: 0.0030 | ds_loss: 0.0000 | lr: 3.1259e-06 | scale: 1.0000 | micro time: 5.293 | step time: 21.083
|
| 121 |
+
train | epoch 4 | Iter: 3996/ 4360 | global iter: 1000/ 1090 | loss: 0.0045 | ds_loss: 0.0000 | lr: 2.1082e-06 | scale: 1.0000 | micro time: 5.286 | step time: 21.088
|
| 122 |
+
train | epoch 4 | Iter: 4076/ 4360 | global iter: 1020/ 1090 | loss: 0.0038 | ds_loss: 0.0000 | lr: 1.2869e-06 | scale: 1.0000 | micro time: 5.291 | step time: 21.085
|
| 123 |
+
train | epoch 4 | Iter: 4156/ 4360 | global iter: 1040/ 1090 | loss: 0.0029 | ds_loss: 0.0000 | lr: 6.6539e-07 | scale: 1.0000 | micro time: 5.286 | step time: 21.085
|
| 124 |
+
train | epoch 4 | Iter: 4236/ 4360 | global iter: 1060/ 1090 | loss: 0.0029 | ds_loss: 0.0000 | lr: 2.4619e-07 | scale: 1.0000 | micro time: 5.285 | step time: 21.083
|
| 125 |
+
train | epoch 4 | Iter: 4316/ 4360 | global iter: 1080/ 1090 | loss: 0.0044 | ds_loss: 0.0000 | lr: 3.1020e-08 | scale: 1.0000 | micro time: 5.289 | step time: 21.087
|
| 126 |
+
dev | avg_loss: 0.38920454545454547 | {'exact_match': 54.1586, 'rougeL': 89.1636}
|