Model save

Browse files

Files changed (13) hide show

README.md +1 -1
all_results.json +3 -3
config.json +5 -6
generation_config.json +10 -2
model-00001-of-00004.safetensors +1 -1
model-00002-of-00004.safetensors +1 -1
model-00003-of-00004.safetensors +1 -1
model-00004-of-00004.safetensors +1 -1
special_tokens_map.json +1 -1
tokenizer_config.json +2 -2
train_results.json +3 -3
trainer_state.json +26 -26
training_args.bin +1 -1

README.md CHANGED Viewed

@@ -26,7 +26,7 @@ print(output["generated_text"])
 ## Training procedure
-[<img src="https://raw.githubusercontent.com/wandb/assets/main/wandb-github-badge-28.svg" alt="Visualize in Weights & Biases" width="150" height="24"/>](https://wandb.ai/1105645918-bit/huggingface/runs/sj64d1o5)
 This model was trained with GRPO, a method introduced in [DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models](https://huggingface.co/papers/2402.03300).

 ## Training procedure
+[<img src="https://raw.githubusercontent.com/wandb/assets/main/wandb-github-badge-28.svg" alt="Visualize in Weights & Biases" width="150" height="24"/>](https://wandb.ai/1105645918-bit/huggingface/runs/2trr7wkk)
 This model was trained with GRPO, a method introduced in [DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models](https://huggingface.co/papers/2402.03300).

all_results.json CHANGED Viewed

@@ -1,8 +1,8 @@
 {
     "total_flos": 0.0,
-    "train_loss": 0.0355598833411932,
-    "train_runtime": 3089.8163,
     "train_samples": 336,
-    "train_samples_per_second": 0.544,
     "train_steps_per_second": 0.003
 }

 {
     "total_flos": 0.0,
+    "train_loss": 0.057270023971796036,
+    "train_runtime": 3066.7214,
     "train_samples": 336,
+    "train_samples_per_second": 0.548,
     "train_steps_per_second": 0.003
 }

config.json CHANGED Viewed

@@ -1,16 +1,16 @@
 {
-  "_name_or_path": "/work/home/liuweichu/Qwen2.5-Math-7B",
   "architectures": [
     "Qwen2ForCausalLM"
   ],
   "attention_dropout": 0.0,
   "bos_token_id": 151643,
-  "eos_token_id": 151643,
   "hidden_act": "silu",
   "hidden_size": 3584,
   "initializer_range": 0.02,
   "intermediate_size": 18944,
-  "max_position_embeddings": 4096,
   "max_window_layers": 28,
   "model_type": "qwen2",
   "num_attention_heads": 28,
@@ -18,13 +18,12 @@
   "num_key_value_heads": 4,
   "rms_norm_eps": 1e-06,
   "rope_scaling": null,
-  "rope_theta": 10000,
-  "sliding_window": 4096,
   "tie_word_embeddings": false,
   "torch_dtype": "bfloat16",
   "transformers_version": "4.49.0",
   "use_cache": false,
-  "use_mrope": false,
   "use_sliding_window": false,
   "vocab_size": 152064
 }

 {
+  "_name_or_path": "/work/home/liuweichu/Qwen2.5-7B-Instruct",
   "architectures": [
     "Qwen2ForCausalLM"
   ],
   "attention_dropout": 0.0,
   "bos_token_id": 151643,
+  "eos_token_id": 151645,
   "hidden_act": "silu",
   "hidden_size": 3584,
   "initializer_range": 0.02,
   "intermediate_size": 18944,
+  "max_position_embeddings": 32768,
   "max_window_layers": 28,
   "model_type": "qwen2",
   "num_attention_heads": 28,
   "num_key_value_heads": 4,
   "rms_norm_eps": 1e-06,
   "rope_scaling": null,
+  "rope_theta": 1000000.0,
+  "sliding_window": 131072,
   "tie_word_embeddings": false,
   "torch_dtype": "bfloat16",
   "transformers_version": "4.49.0",
   "use_cache": false,
   "use_sliding_window": false,
   "vocab_size": 152064
 }

generation_config.json CHANGED Viewed

@@ -1,6 +1,14 @@
 {
   "bos_token_id": 151643,
-  "eos_token_id": 151643,
-  "max_new_tokens": 2048,
   "transformers_version": "4.49.0"
 }

 {
   "bos_token_id": 151643,
+  "do_sample": true,
+  "eos_token_id": [
+    151645,
+    151643
+  ],
+  "pad_token_id": 151643,
+  "repetition_penalty": 1.05,
+  "temperature": 0.7,
+  "top_k": 20,
+  "top_p": 0.8,
   "transformers_version": "4.49.0"
 }

model-00001-of-00004.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:6f010c93aeb6a90a25aa922385f63985ac85ec90d512fcd5cf291ac497dbef47
 size 4877660776

 version https://git-lfs.github.com/spec/v1
+oid sha256:d695ba33451eb7a1bf201982146c72f5315781c4edb0880d441f804363bafbdf
 size 4877660776

model-00002-of-00004.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:d9fb7055b43b9f115ab3bec564a91d13866a9b1f83ceea207ac222bcd8a81e6d
 size 4932751008

 version https://git-lfs.github.com/spec/v1
+oid sha256:240bc2c3be7f25610723478c3fba8cfc7ffea89000dca5c83e33896010ac2bf6
 size 4932751008

model-00003-of-00004.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:f4b53cc05267704569932632f11059e8c854ce0e88a11d3bb700ecf8689028ee
 size 4330865200

 version https://git-lfs.github.com/spec/v1
+oid sha256:1c2d798bae01b93ef2439b1b1868076e8ce331883d5be9ea20543af984498df8
 size 4330865200

model-00004-of-00004.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:36db9149ef4720ed8f11de88e85145219577fb349a8c3dab8bcc95305b6f48cd
 size 1089994880

 version https://git-lfs.github.com/spec/v1
+oid sha256:3b23d0800c8a6b0fc258da6ed43a1df4e253dcec06c0b9ff79c4c6bb56286b7d
 size 1089994880

special_tokens_map.json CHANGED Viewed

@@ -15,7 +15,7 @@
     "<|video_pad|>"
   ],
   "eos_token": {
-    "content": "<|endoftext|>",
     "lstrip": false,
     "normalized": false,
     "rstrip": false,

     "<|video_pad|>"
   ],
   "eos_token": {
+    "content": "<|im_end|>",
     "lstrip": false,
     "normalized": false,
     "rstrip": false,

tokenizer_config.json CHANGED Viewed

@@ -195,9 +195,9 @@
     "<|video_pad|>"
   ],
   "bos_token": null,
-  "chat_template": "{%- if tools %}\n    {{- '<|im_start|>system\\n' }}\n    {%- if messages[0]['role'] == 'system' %}\n        {{- messages[0]['content'] }}\n    {%- else %}\n        {{- 'Please reason step by step, and put your final answer within \\\\boxed{}.' }}\n    {%- endif %}\n    {{- \"\\n\\n# Tools\\n\\nYou may call one or more functions to assist with the user query.\\n\\nYou are provided with function signatures within <tools></tools> XML tags:\\n<tools>\" }}\n    {%- for tool in tools %}\n        {{- \"\\n\" }}\n        {{- tool | tojson }}\n    {%- endfor %}\n    {{- \"\\n</tools>\\n\\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\\n<tool_call>\\n{\\\"name\\\": <function-name>, \\\"arguments\\\": <args-json-object>}\\n</tool_call><|im_end|>\\n\" }}\n{%- else %}\n    {%- if messages[0]['role'] == 'system' %}\n        {{- '<|im_start|>system\\n' + messages[0]['content'] + '<|im_end|>\\n' }}\n    {%- else %}\n        {{- '<|im_start|>system\\nPlease reason step by step, and put your final answer within \\\\boxed{}.<|im_end|>\\n' }}\n    {%- endif %}\n{%- endif %}\n{%- for message in messages %}\n    {%- if (message.role == \"user\") or (message.role == \"system\" and not loop.first) or (message.role == \"assistant\" and not message.tool_calls) %}\n        {{- '<|im_start|>' + message.role + '\\n' + message.content + '<|im_end|>' + '\\n' }}\n    {%- elif message.role == \"assistant\" %}\n        {{- '<|im_start|>' + message.role }}\n        {%- if message.content %}\n            {{- '\\n' + message.content }}\n        {%- endif %}\n        {%- for tool_call in message.tool_calls %}\n            {%- if tool_call.function is defined %}\n                {%- set tool_call = tool_call.function %}\n            {%- endif %}\n            {{- '\\n<tool_call>\\n{\"name\": \"' }}\n            {{- tool_call.name }}\n            {{- '\", \"arguments\": ' }}\n            {{- tool_call.arguments | tojson }}\n            {{- '}\\n</tool_call>' }}\n        {%- endfor %}\n        {{- '<|im_end|>\\n' }}\n    {%- elif message.role == \"tool\" %}\n        {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != \"tool\") %}\n            {{- '<|im_start|>user' }}\n        {%- endif %}\n        {{- '\\n<tool_response>\\n' }}\n        {{- message.content }}\n        {{- '\\n</tool_response>' }}\n        {%- if loop.last or (messages[loop.index0 + 1].role != \"tool\") %}\n            {{- '<|im_end|>\\n' }}\n        {%- endif %}\n    {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n    {{- '<|im_start|>assistant\\n' }}\n{%- endif %}\n",
   "clean_up_tokenization_spaces": false,
-  "eos_token": "<|endoftext|>",
   "errors": "replace",
   "extra_special_tokens": {},
   "model_max_length": 131072,

     "<|video_pad|>"
   ],
   "bos_token": null,
+  "chat_template": "{%- if tools %}\n    {{- '<|im_start|>system\\n' }}\n    {%- if messages[0]['role'] == 'system' %}\n        {{- messages[0]['content'] }}\n    {%- else %}\n        {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}\n    {%- endif %}\n    {{- \"\\n\\n# Tools\\n\\nYou may call one or more functions to assist with the user query.\\n\\nYou are provided with function signatures within <tools></tools> XML tags:\\n<tools>\" }}\n    {%- for tool in tools %}\n        {{- \"\\n\" }}\n        {{- tool | tojson }}\n    {%- endfor %}\n    {{- \"\\n</tools>\\n\\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\\n<tool_call>\\n{\\\"name\\\": <function-name>, \\\"arguments\\\": <args-json-object>}\\n</tool_call><|im_end|>\\n\" }}\n{%- else %}\n    {%- if messages[0]['role'] == 'system' %}\n        {{- '<|im_start|>system\\n' + messages[0]['content'] + '<|im_end|>\\n' }}\n    {%- else %}\n        {{- '<|im_start|>system\\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\\n' }}\n    {%- endif %}\n{%- endif %}\n{%- for message in messages %}\n    {%- if (message.role == \"user\") or (message.role == \"system\" and not loop.first) or (message.role == \"assistant\" and not message.tool_calls) %}\n        {{- '<|im_start|>' + message.role + '\\n' + message.content + '<|im_end|>' + '\\n' }}\n    {%- elif message.role == \"assistant\" %}\n        {{- '<|im_start|>' + message.role }}\n        {%- if message.content %}\n            {{- '\\n' + message.content }}\n        {%- endif %}\n        {%- for tool_call in message.tool_calls %}\n            {%- if tool_call.function is defined %}\n                {%- set tool_call = tool_call.function %}\n            {%- endif %}\n            {{- '\\n<tool_call>\\n{\"name\": \"' }}\n            {{- tool_call.name }}\n            {{- '\", \"arguments\": ' }}\n            {{- tool_call.arguments | tojson }}\n            {{- '}\\n</tool_call>' }}\n        {%- endfor %}\n        {{- '<|im_end|>\\n' }}\n    {%- elif message.role == \"tool\" %}\n        {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != \"tool\") %}\n            {{- '<|im_start|>user' }}\n        {%- endif %}\n        {{- '\\n<tool_response>\\n' }}\n        {{- message.content }}\n        {{- '\\n</tool_response>' }}\n        {%- if loop.last or (messages[loop.index0 + 1].role != \"tool\") %}\n            {{- '<|im_end|>\\n' }}\n        {%- endif %}\n    {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n    {{- '<|im_start|>assistant\\n' }}\n{%- endif %}\n",
   "clean_up_tokenization_spaces": false,
+  "eos_token": "<|im_end|>",
   "errors": "replace",
   "extra_special_tokens": {},
   "model_max_length": 131072,

train_results.json CHANGED Viewed

@@ -1,8 +1,8 @@
 {
     "total_flos": 0.0,
-    "train_loss": 0.0355598833411932,
-    "train_runtime": 3089.8163,
     "train_samples": 336,
-    "train_samples_per_second": 0.544,
     "train_steps_per_second": 0.003
 }

 {
     "total_flos": 0.0,
+    "train_loss": 0.057270023971796036,
+    "train_runtime": 3066.7214,
     "train_samples": 336,
+    "train_samples_per_second": 0.548,
     "train_steps_per_second": 0.003
 }

trainer_state.json CHANGED Viewed

@@ -10,53 +10,53 @@
   "log_history": [
     {
       "clip_ratio": 0.0,
-      "completion_length": 760.5469055175781,
       "epoch": 0.38095238095238093,
-      "grad_norm": 0.19067028164863586,
       "kl": 0.0,
       "learning_rate": 3e-06,
-      "loss": 0.0244,
-      "reward": 0.2343750111758709,
-      "reward_std": 0.2009137775748968,
-      "rewards/accuracy_reward": 0.19754465017467737,
-      "rewards/format_reward": 0.036830359254963696,
       "step": 1
     },
     {
       "clip_ratio": 0.0,
-      "completion_length": 759.1691207885742,
       "epoch": 2.380952380952381,
-      "grad_norm": 0.47505804896354675,
-      "kl": 0.004555165767669678,
       "learning_rate": 1.7604722665003958e-06,
-      "loss": 0.0261,
-      "reward": 0.2315848316065967,
-      "reward_std": 0.25816529244184494,
-      "rewards/accuracy_reward": 0.18470982974395156,
-      "rewards/format_reward": 0.046875002240994945,
       "step": 5
     },
     {
       "clip_ratio": 0.0,
-      "completion_length": 740.1288314819336,
       "epoch": 4.761904761904762,
-      "grad_norm": 0.3742629885673523,
-      "kl": 0.022109222412109376,
       "learning_rate": 0.0,
-      "loss": 0.0453,
-      "reward": 0.30781251527369025,
-      "reward_std": 0.34542269371449946,
-      "rewards/accuracy_reward": 0.18437500880099833,
-      "rewards/format_reward": 0.12343750623986124,
       "step": 10
     },
     {
       "epoch": 4.761904761904762,
       "step": 10,
       "total_flos": 0.0,
-      "train_loss": 0.0355598833411932,
-      "train_runtime": 3089.8163,
-      "train_samples_per_second": 0.544,
       "train_steps_per_second": 0.003
     }
   ],

   "log_history": [
     {
       "clip_ratio": 0.0,
+      "completion_length": 999.6105346679688,
       "epoch": 0.38095238095238093,
+      "grad_norm": 154.4364013671875,
       "kl": 0.0,
       "learning_rate": 3e-06,
+      "loss": -0.019,
+      "reward": 0.7500000298023224,
+      "reward_std": 0.3680399917066097,
+      "rewards/accuracy_reward": 0.031250001629814506,
+      "rewards/format_reward": 0.7187500298023224,
       "step": 1
     },
     {
       "clip_ratio": 0.0,
+      "completion_length": 1006.7606468200684,
       "epoch": 2.380952380952381,
+      "grad_norm": 8.609855651855469,
+      "kl": 1.3177490234375,
       "learning_rate": 1.7604722665003958e-06,
+      "loss": 0.0393,
+      "reward": 0.7695312835276127,
+      "reward_std": 0.3738137981854379,
+      "rewards/accuracy_reward": 0.04101562706637196,
+      "rewards/format_reward": 0.7285156585276127,
       "step": 5
     },
     {
       "clip_ratio": 0.0,
+      "completion_length": 1006.158299255371,
       "epoch": 4.761904761904762,
+      "grad_norm": 0.9559445381164551,
+      "kl": 2.4314453125,
       "learning_rate": 0.0,
+      "loss": 0.0869,
+      "reward": 0.7937500372529029,
+      "reward_std": 0.36111804023385047,
+      "rewards/accuracy_reward": 0.03906250209547579,
+      "rewards/format_reward": 0.7546875327825546,
       "step": 10
     },
     {
       "epoch": 4.761904761904762,
       "step": 10,
       "total_flos": 0.0,
+      "train_loss": 0.057270023971796036,
+      "train_runtime": 3066.7214,
+      "train_samples_per_second": 0.548,
       "train_steps_per_second": 0.003
     }
   ],

training_args.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:08f748b3c516d433cd88ac282eefed6ff303d83a3f16ccb32889ff32d0bccdec
 size 8120

 version https://git-lfs.github.com/spec/v1
+oid sha256:2208791588500d187944b80bb13411e7e17df3477815ff62483411ddd5abe7a7
 size 8120