mudler commited on 3 days ago

Commit

dac778a

verified ·

1 Parent(s): d772624

Upload finetuned FunctionGemma for Italian function calling

Browse files

This view is limited to 50 files because it contains too many changes. See raw diff

Files changed (50) hide show

.gitattributes +5 -0
README.md +58 -0
chat_template.jinja +279 -0
checkpoint-154/chat_template.jinja +279 -0
checkpoint-154/config.json +62 -0
checkpoint-154/generation_config.json +15 -0
checkpoint-154/model.safetensors +3 -0
checkpoint-154/optimizer.pt +3 -0
checkpoint-154/rng_state.pth +3 -0
checkpoint-154/scheduler.pt +3 -0
checkpoint-154/tokenizer.json +3 -0
checkpoint-154/tokenizer_config.json +26 -0
checkpoint-154/trainer_state.json +97 -0
checkpoint-154/training_args.bin +3 -0
checkpoint-231/chat_template.jinja +279 -0
checkpoint-231/config.json +62 -0
checkpoint-231/generation_config.json +15 -0
checkpoint-231/model.safetensors +3 -0
checkpoint-231/optimizer.pt +3 -0
checkpoint-231/rng_state.pth +3 -0
checkpoint-231/scheduler.pt +3 -0
checkpoint-231/tokenizer.json +3 -0
checkpoint-231/tokenizer_config.json +26 -0
checkpoint-231/trainer_state.json +118 -0
checkpoint-231/training_args.bin +3 -0
checkpoint-308/chat_template.jinja +279 -0
checkpoint-308/config.json +62 -0
checkpoint-308/generation_config.json +15 -0
checkpoint-308/model.safetensors +3 -0
checkpoint-308/optimizer.pt +3 -0
checkpoint-308/rng_state.pth +3 -0
checkpoint-308/scheduler.pt +3 -0
checkpoint-308/tokenizer.json +3 -0
checkpoint-308/tokenizer_config.json +26 -0
checkpoint-308/trainer_state.json +171 -0
checkpoint-308/training_args.bin +3 -0
checkpoint-77/chat_template.jinja +279 -0
checkpoint-77/config.json +62 -0
checkpoint-77/generation_config.json +15 -0
checkpoint-77/model.safetensors +3 -0
checkpoint-77/optimizer.pt +3 -0
checkpoint-77/rng_state.pth +3 -0
checkpoint-77/scheduler.pt +3 -0
checkpoint-77/tokenizer.json +3 -0
checkpoint-77/tokenizer_config.json +26 -0
checkpoint-77/trainer_state.json +55 -0
checkpoint-77/training_args.bin +3 -0
config.json +62 -0
generation_config.json +15 -0
model.safetensors +3 -0

.gitattributes CHANGED Viewed

@@ -33,3 +33,8 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+checkpoint-154/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-231/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-308/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-77/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+tokenizer.json filter=lfs diff=lfs merge=lfs -text

README.md ADDED Viewed

	@@ -0,0 +1,58 @@

+---
+base_model: google/functiongemma-270m-it
+library_name: transformers
+model_name: outputs
+tags:
+- generated_from_trainer
+- sft
+- trl
+licence: license
+---
+# Model Card for outputs
+This model is a fine-tuned version of [google/functiongemma-270m-it](https://huggingface.co/google/functiongemma-270m-it).
+It has been trained using [TRL](https://github.com/huggingface/trl).
+## Quick start
+```python
+from transformers import pipeline
+question = "If you had a time machine, but could only go to the past or the future once and never return, which would you choose and why?"
+generator = pipeline("text-generation", model="None", device="cuda")
+output = generator([{"role": "user", "content": question}], max_new_tokens=128, return_full_text=False)[0]
+print(output["generated_text"])
+```
+## Training procedure
+This model was trained with SFT.
+### Framework versions
+- TRL: 1.0.0
+- Transformers: 5.5.1
+- Pytorch: 2.11.0
+- Datasets: 4.8.4
+- Tokenizers: 0.22.2
+## Citations
+Cite TRL as:
+```bibtex
+@software{vonwerra2020trl,
+  title   = {{TRL: Transformers Reinforcement Learning}},
+  author  = {von Werra, Leandro and Belkada, Younes and Tunstall, Lewis and Beeching, Edward and Thrush, Tristan and Lambert, Nathan and Huang, Shengyi and Rasul, Kashif and Gallouédec, Quentin},
+  license = {Apache-2.0},
+  url     = {https://github.com/huggingface/trl},
+  year    = {2020}
+}
+```

chat_template.jinja ADDED Viewed

	@@ -0,0 +1,279 @@

+{%- macro format_parameters(properties, required) -%}
+    {%- set standard_keys = ['description', 'type', 'properties', 'required', 'nullable'] -%}
+    {%- set ns = namespace(found_first=false) -%}
+    {%- for key, value in properties | dictsort -%}
+        {%- if key not in standard_keys -%}
+            {%- if ns.found_first %},{% endif -%}
+            {%- set ns.found_first = true -%}
+            {{- key }}:{description:<escape>{{ value['description'] }}<escape>
+            {%- if value['type'] | upper == 'STRING' -%}
+                {%- if value['enum'] -%}
+                    ,enum:{{ format_argument(value['enum']) }}
+                {%- endif -%}
+            {%- elif value['type'] | upper == 'OBJECT' -%}
+                ,properties:{
+                {%- if value['properties'] is defined and value['properties'] is mapping -%}
+                    {{- format_parameters(value['properties'], value['required'] | default([])) -}}
+                {%- elif value is mapping -%}
+                    {{- format_parameters(value, value['required'] | default([])) -}}
+                {%- endif -%}
+                }
+                {%- if value['required'] -%}
+                    ,required:[
+                    {%- for item in value['required'] | default([]) -%}
+                        <escape>{{- item -}}<escape>
+                        {%- if not loop.last %},{% endif -%}
+                    {%- endfor -%}
+                    ]
+                {%- endif -%}
+            {%- elif value['type'] | upper == 'ARRAY' -%}
+                {%- if value['items'] is mapping and value['items'] -%}
+                    ,items:{
+                    {%- set ns_items = namespace(found_first=false) -%}
+                    {%- for item_key, item_value in value['items'] | dictsort -%}
+                        {%- if item_value is not none -%}
+                            {%- if ns_items.found_first %},{% endif -%}
+                            {%- set ns_items.found_first = true -%}
+                            {%- if item_key == 'properties' -%}
+                                properties:{
+                                {%- if item_value is mapping -%}
+                                    {{- format_parameters(item_value, value['items']['required'] | default([])) -}}
+                                {%- endif -%}
+                                }
+                            {%- elif item_key == 'required' -%}
+                                required:[
+                                {%- for req_item in item_value -%}
+                                    <escape>{{- req_item -}}<escape>
+                                    {%- if not loop.last %},{% endif -%}
+                                {%- endfor -%}
+                                ]
+                            {%- elif item_key == 'type' -%}
+                                {%- if item_value is string -%}
+                                    type:{{ format_argument(item_value | upper) }}
+                                {%- else -%}
+                                    type:{{ format_argument(item_value | map('upper') | list) }}
+                                {%- endif -%}
+                            {%- else -%}
+                                {{ item_key }}:{{ format_argument(item_value) }}
+                            {%- endif -%}
+                        {%- endif -%}
+                    {%- endfor -%}
+                    }
+                {%- endif -%}
+            {%- endif -%}
+            ,type:<escape>{{ value['type'] | upper }}<escape>}
+        {%- endif -%}
+    {%- endfor -%}
+{%- endmacro -%}
+{% macro format_function_declaration(tool_data) -%}
+declaration:{{- tool_data['function']['name'] -}}
+{description:<escape>{{- tool_data['function']['description'] -}}<escape>
+{%- set params = tool_data['function']['parameters'] -%}
+{%- if params -%}
+    ,parameters:{
+    {%- if params['properties'] -%}
+        properties:{ {{- format_parameters(params['properties'], params['required']) -}} },
+    {%- endif -%}
+    {%- if params['required'] -%}
+        required:[
+        {%- for item in params['required'] -%}
+            <escape>{{- item -}}<escape>
+            {{- ',' if not loop.last -}}
+        {%- endfor -%}
+        ],
+    {%- endif -%}
+    {%- if params['type'] -%}
+        type:<escape>{{- params['type'] | upper -}}<escape>}
+    {%- endif -%}
+{%- endif -%}
+}
+{%- endmacro -%}
+{% macro format_argument(argument, escape_keys=True) -%}
+{%- if argument is string -%}
+    {{- '<escape>' + argument + '<escape>' -}}
+{%- elif argument is boolean -%}
+    {%- if argument -%}
+        {{- 'true' -}}
+    {%- else -%}
+        {{- 'false' -}}
+    {%- endif -%}
+{%- elif argument is mapping -%}
+    {{- '{' -}}
+    {%- set ns = namespace(found_first=false) -%}
+    {%- for key, value in argument | dictsort -%}
+        {%- if ns.found_first %},{% endif -%}
+        {%- set ns.found_first = true -%}
+        {%- if escape_keys -%}
+            {{- '<escape>' + key + '<escape>' -}}
+        {%- else -%}
+            {{- key -}}
+        {%- endif -%}
+        :{{- format_argument(value, escape_keys=escape_keys) -}}
+    {%- endfor -%}
+    {{- '}' -}}
+{%- elif argument is sequence -%}
+    {{- '[' -}}
+    {%- for item in argument -%}
+        {{- format_argument(item, escape_keys=escape_keys) -}}
+        {%- if not loop.last %},{% endif -%}
+    {%- endfor -%}
+    {{- ']' -}}
+{%- else -%}
+    {{- argument -}}
+{%- endif -%}
+{%- endmacro -%}
+{{ bos_token }}
+{%- set ns = namespace(prev_message_type=None) -%}
+{#- Tool Declarations -#}
+{%- set loop_messages = messages -%}
+{%- if tools or messages[0]['role'] == 'system' or messages[0]['role'] == 'developer' -%}
+    {{- '<start_of_turn>developer\n' -}}
+    {%- if messages[0]['role'] == 'system' or messages[0]['role'] == 'developer' -%}
+        {%- if messages[0]['content'] is string -%}
+            {{- messages[0]['content'] | trim -}}
+        {%- elif messages[0]['content'] is sequence -%}
+            {%- for item in messages[0]['content'] -%}
+                {%- if item['type'] == 'text' -%}
+                    {{- item['text'] | trim -}}
+                {%- endif -%}
+            {%- endfor -%}
+        {%- endif -%}
+        {%- set loop_messages = messages[1:] -%}
+    {%- endif -%}
+    {%- if tools -%}
+        {%- for tool in tools %}
+            {{- '<start_function_declaration>' -}}
+            {{- format_function_declaration(tool) | trim }}
+            {{- '<end_function_declaration>' -}}
+        {%- endfor %}
+    {%- endif -%}
+    {{- '<end_of_turn>\n' }}
+{%- endif %}
+{#- Loop through messages. -#}
+{%- for message in loop_messages -%}
+    {%- if (message['role'] == 'assistant') -%}
+        {#- Rename "assistant" to "model". -#}
+        {%- set role = "model" -%}
+    {%- else -%}
+        {%- set role = message['role'] -%}
+    {%- endif -%}
+    {%- if role != 'tool' -%}
+        {%- if ns.prev_message_type != 'tool_response' -%}
+            {{- '<start_of_turn>' + role + '\n' }}
+        {%- endif -%}
+        {%- set ns.prev_message_type = None -%}
+        {%- if 'content' in message and message['content'] is not none -%}
+            {%- if message['content'] is string -%}
+                {{ message['content'] | trim }}
+            {%- elif message['content'] is sequence -%}
+                {%- for item in message['content'] -%}
+                    {%- if item['type'] == 'image' -%}
+                        {{ '<start_of_image>' }}
+                    {%- elif item['type'] == 'text' -%}
+                        {{ item['text'] | trim }}
+                    {%- endif -%}
+                {%- endfor -%}
+            {%- else -%}
+                {{ raise_exception("Invalid content type in user/assistant message") }}
+            {%- endif -%}
+            {%- set ns.prev_message_type = 'content' -%}
+        {%- endif -%}
+        {%- if 'tool_calls' in message and message['tool_calls'] and message['tool_calls'] is iterable -%}
+            {#- Tool Calls -#}
+            {%- for tool_call in message['tool_calls'] -%}
+                {% set function = tool_call['function'] %}
+                {{-  '<start_function_call>call:' + function['name'] + '{' -}}
+                {%- if 'arguments' in function -%}
+                    {%- if function['arguments'] is mapping -%}
+                        {%- set ns = namespace(found_first=false) -%}
+                        {%- for key, value in function['arguments'] | dictsort -%}
+                            {%- if ns.found_first %},{% endif -%}
+                            {%- set ns.found_first = true -%}
+                            {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                        {%- endfor -%}
+                    {%- elif function['arguments'] is string -%}
+                        {# This handles string-JSON, just in case #}
+                    {{ function['arguments'] }}
+                    {%- endif %}
+                {%- endif -%}
+                {{- '}<end_function_call>' -}}
+            {%- endfor -%}
+            {%- if loop.last -%}
+                {{ '<start_function_response>' }}
+            {%- endif -%}
+            {%- set ns.prev_message_type = 'tool_call' -%}
+        {%- endif -%}
+    {%- else -%}
+        {#- Tool Responses -#}
+        {%- if 'content' in message and message['content'] -%}
+            {%- if message['content'] is mapping -%}
+                {%- if 'name' in message['content'] and 'response' in message['content'] -%}
+                    {{ '<start_function_response>response:' + message['content']['name'] | trim + '{' }}
+                    {%- set response_ns = namespace(found_first=false) -%}
+                    {%- for key, value in message['content']['response'] | dictsort -%}
+                        {%- if response_ns.found_first %},{% endif -%}
+                        {%- set response_ns.found_first = true -%}
+                        {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                    {%- endfor -%}
+                    {{- '}<end_function_response>' -}}
+                {%- elif 'name' in message -%}
+                    {{ '<start_function_response>response:' + message['name'] | trim + '{' }}
+                    {%- set response_ns = namespace(found_first=false) -%}
+                    {%- for key, value in message['content'] | dictsort -%}
+                        {%- if response_ns.found_first %},{% endif -%}
+                        {%- set response_ns.found_first = true -%}
+                        {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                    {%- endfor -%}
+                    {{- '}<end_function_response>' -}}
+                {%- else -%}
+                    {{ raise_exception("Invalid tool response mapping: must contain 'name' and 'response' keys, or 'name' must be in the message.") }}
+                {%- endif -%}
+            {%- elif message['content'] is string -%}
+                {%- if 'name' in message -%}
+                     {{ '<start_function_response>response:' + message['name'] | trim + '{value:' + format_argument(message['content'], escape_keys=False) + '}<end_function_response>' }}
+                {%- else -%}
+                     {{ raise_exception("Invalid tool response: 'name' must be provided.") }}
+                {%- endif -%}
+            {%- elif message['content'] is sequence -%}
+                {%- for item in message['content'] -%}
+                    {%- if item is mapping -%}
+                        {%- if 'name' in item and 'response' in item -%}
+                            {{ '<start_function_response>response:' + item['name'] | trim + '{' }}
+                            {%- set response_ns = namespace(found_first=false) -%}
+                            {%- for key, value in item['response'] | dictsort -%}
+                                {%- if response_ns.found_first %},{% endif -%}
+                                {%- set response_ns.found_first = true -%}
+                                {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                            {%- endfor -%}
+                            {{- '}<end_function_response>' -}}
+                        {%- elif 'name' in message -%}
+                            {{ '<start_function_response>response:' + message['name'] | trim + '{' }}
+                            {%- set response_ns = namespace(found_first=false) -%}
+                            {%- for key, value in item | dictsort -%}
+                                {%- if response_ns.found_first %},{% endif -%}
+                                {%- set response_ns.found_first = true -%}
+                                {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                            {%- endfor -%}
+                            {{- '}<end_function_response>' -}}
+                        {%- else -%}
+                            {{ raise_exception("Invalid tool response mapping: must contain 'name' and 'response' keys, or 'name' must be in the message.") }}
+                        {%- endif -%}
+                    {%- else -%}
+                        {{ raise_exception("Invalid tool response message: multiple responses must all be mappings") }}
+                    {%- endif -%}
+                {%- endfor -%}
+            {%- else -%}
+                {{ raise_exception("Invalid content type in tool message: must be mapping, sequence of mappings, or string.") }}
+            {%- endif -%}
+        {%- endif -%}
+        {%- set ns.prev_message_type = 'tool_response' -%}
+    {%- endif -%}
+    {%- if ns.prev_message_type not in ['tool_call', 'tool_response'] -%}
+        {{ '<end_of_turn>\n' }}
+    {%- endif -%}
+{%- endfor -%}
+{%- if add_generation_prompt -%}
+    {%- if ns.prev_message_type != 'tool_response' -%}
+        {{- '<start_of_turn>model\n' -}}
+    {%- endif -%}
+{%- endif -%}

checkpoint-154/chat_template.jinja ADDED Viewed

	@@ -0,0 +1,279 @@

+{%- macro format_parameters(properties, required) -%}
+    {%- set standard_keys = ['description', 'type', 'properties', 'required', 'nullable'] -%}
+    {%- set ns = namespace(found_first=false) -%}
+    {%- for key, value in properties | dictsort -%}
+        {%- if key not in standard_keys -%}
+            {%- if ns.found_first %},{% endif -%}
+            {%- set ns.found_first = true -%}
+            {{- key }}:{description:<escape>{{ value['description'] }}<escape>
+            {%- if value['type'] | upper == 'STRING' -%}
+                {%- if value['enum'] -%}
+                    ,enum:{{ format_argument(value['enum']) }}
+                {%- endif -%}
+            {%- elif value['type'] | upper == 'OBJECT' -%}
+                ,properties:{
+                {%- if value['properties'] is defined and value['properties'] is mapping -%}
+                    {{- format_parameters(value['properties'], value['required'] | default([])) -}}
+                {%- elif value is mapping -%}
+                    {{- format_parameters(value, value['required'] | default([])) -}}
+                {%- endif -%}
+                }
+                {%- if value['required'] -%}
+                    ,required:[
+                    {%- for item in value['required'] | default([]) -%}
+                        <escape>{{- item -}}<escape>
+                        {%- if not loop.last %},{% endif -%}
+                    {%- endfor -%}
+                    ]
+                {%- endif -%}
+            {%- elif value['type'] | upper == 'ARRAY' -%}
+                {%- if value['items'] is mapping and value['items'] -%}
+                    ,items:{
+                    {%- set ns_items = namespace(found_first=false) -%}
+                    {%- for item_key, item_value in value['items'] | dictsort -%}
+                        {%- if item_value is not none -%}
+                            {%- if ns_items.found_first %},{% endif -%}
+                            {%- set ns_items.found_first = true -%}
+                            {%- if item_key == 'properties' -%}
+                                properties:{
+                                {%- if item_value is mapping -%}
+                                    {{- format_parameters(item_value, value['items']['required'] | default([])) -}}
+                                {%- endif -%}
+                                }
+                            {%- elif item_key == 'required' -%}
+                                required:[
+                                {%- for req_item in item_value -%}
+                                    <escape>{{- req_item -}}<escape>
+                                    {%- if not loop.last %},{% endif -%}
+                                {%- endfor -%}
+                                ]
+                            {%- elif item_key == 'type' -%}
+                                {%- if item_value is string -%}
+                                    type:{{ format_argument(item_value | upper) }}
+                                {%- else -%}
+                                    type:{{ format_argument(item_value | map('upper') | list) }}
+                                {%- endif -%}
+                            {%- else -%}
+                                {{ item_key }}:{{ format_argument(item_value) }}
+                            {%- endif -%}
+                        {%- endif -%}
+                    {%- endfor -%}
+                    }
+                {%- endif -%}
+            {%- endif -%}
+            ,type:<escape>{{ value['type'] | upper }}<escape>}
+        {%- endif -%}
+    {%- endfor -%}
+{%- endmacro -%}
+{% macro format_function_declaration(tool_data) -%}
+declaration:{{- tool_data['function']['name'] -}}
+{description:<escape>{{- tool_data['function']['description'] -}}<escape>
+{%- set params = tool_data['function']['parameters'] -%}
+{%- if params -%}
+    ,parameters:{
+    {%- if params['properties'] -%}
+        properties:{ {{- format_parameters(params['properties'], params['required']) -}} },
+    {%- endif -%}
+    {%- if params['required'] -%}
+        required:[
+        {%- for item in params['required'] -%}
+            <escape>{{- item -}}<escape>
+            {{- ',' if not loop.last -}}
+        {%- endfor -%}
+        ],
+    {%- endif -%}
+    {%- if params['type'] -%}
+        type:<escape>{{- params['type'] | upper -}}<escape>}
+    {%- endif -%}
+{%- endif -%}
+}
+{%- endmacro -%}
+{% macro format_argument(argument, escape_keys=True) -%}
+{%- if argument is string -%}
+    {{- '<escape>' + argument + '<escape>' -}}
+{%- elif argument is boolean -%}
+    {%- if argument -%}
+        {{- 'true' -}}
+    {%- else -%}
+        {{- 'false' -}}
+    {%- endif -%}
+{%- elif argument is mapping -%}
+    {{- '{' -}}
+    {%- set ns = namespace(found_first=false) -%}
+    {%- for key, value in argument | dictsort -%}
+        {%- if ns.found_first %},{% endif -%}
+        {%- set ns.found_first = true -%}
+        {%- if escape_keys -%}
+            {{- '<escape>' + key + '<escape>' -}}
+        {%- else -%}
+            {{- key -}}
+        {%- endif -%}
+        :{{- format_argument(value, escape_keys=escape_keys) -}}
+    {%- endfor -%}
+    {{- '}' -}}
+{%- elif argument is sequence -%}
+    {{- '[' -}}
+    {%- for item in argument -%}
+        {{- format_argument(item, escape_keys=escape_keys) -}}
+        {%- if not loop.last %},{% endif -%}
+    {%- endfor -%}
+    {{- ']' -}}
+{%- else -%}
+    {{- argument -}}
+{%- endif -%}
+{%- endmacro -%}
+{{ bos_token }}
+{%- set ns = namespace(prev_message_type=None) -%}
+{#- Tool Declarations -#}
+{%- set loop_messages = messages -%}
+{%- if tools or messages[0]['role'] == 'system' or messages[0]['role'] == 'developer' -%}
+    {{- '<start_of_turn>developer\n' -}}
+    {%- if messages[0]['role'] == 'system' or messages[0]['role'] == 'developer' -%}
+        {%- if messages[0]['content'] is string -%}
+            {{- messages[0]['content'] | trim -}}
+        {%- elif messages[0]['content'] is sequence -%}
+            {%- for item in messages[0]['content'] -%}
+                {%- if item['type'] == 'text' -%}
+                    {{- item['text'] | trim -}}
+                {%- endif -%}
+            {%- endfor -%}
+        {%- endif -%}
+        {%- set loop_messages = messages[1:] -%}
+    {%- endif -%}
+    {%- if tools -%}
+        {%- for tool in tools %}
+            {{- '<start_function_declaration>' -}}
+            {{- format_function_declaration(tool) | trim }}
+            {{- '<end_function_declaration>' -}}
+        {%- endfor %}
+    {%- endif -%}
+    {{- '<end_of_turn>\n' }}
+{%- endif %}
+{#- Loop through messages. -#}
+{%- for message in loop_messages -%}
+    {%- if (message['role'] == 'assistant') -%}
+        {#- Rename "assistant" to "model". -#}
+        {%- set role = "model" -%}
+    {%- else -%}
+        {%- set role = message['role'] -%}
+    {%- endif -%}
+    {%- if role != 'tool' -%}
+        {%- if ns.prev_message_type != 'tool_response' -%}
+            {{- '<start_of_turn>' + role + '\n' }}
+        {%- endif -%}
+        {%- set ns.prev_message_type = None -%}
+        {%- if 'content' in message and message['content'] is not none -%}
+            {%- if message['content'] is string -%}
+                {{ message['content'] | trim }}
+            {%- elif message['content'] is sequence -%}
+                {%- for item in message['content'] -%}
+                    {%- if item['type'] == 'image' -%}
+                        {{ '<start_of_image>' }}
+                    {%- elif item['type'] == 'text' -%}
+                        {{ item['text'] | trim }}
+                    {%- endif -%}
+                {%- endfor -%}
+            {%- else -%}
+                {{ raise_exception("Invalid content type in user/assistant message") }}
+            {%- endif -%}
+            {%- set ns.prev_message_type = 'content' -%}
+        {%- endif -%}
+        {%- if 'tool_calls' in message and message['tool_calls'] and message['tool_calls'] is iterable -%}
+            {#- Tool Calls -#}
+            {%- for tool_call in message['tool_calls'] -%}
+                {% set function = tool_call['function'] %}
+                {{-  '<start_function_call>call:' + function['name'] + '{' -}}
+                {%- if 'arguments' in function -%}
+                    {%- if function['arguments'] is mapping -%}
+                        {%- set ns = namespace(found_first=false) -%}
+                        {%- for key, value in function['arguments'] | dictsort -%}
+                            {%- if ns.found_first %},{% endif -%}
+                            {%- set ns.found_first = true -%}
+                            {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                        {%- endfor -%}
+                    {%- elif function['arguments'] is string -%}
+                        {# This handles string-JSON, just in case #}
+                    {{ function['arguments'] }}
+                    {%- endif %}
+                {%- endif -%}
+                {{- '}<end_function_call>' -}}
+            {%- endfor -%}
+            {%- if loop.last -%}
+                {{ '<start_function_response>' }}
+            {%- endif -%}
+            {%- set ns.prev_message_type = 'tool_call' -%}
+        {%- endif -%}
+    {%- else -%}
+        {#- Tool Responses -#}
+        {%- if 'content' in message and message['content'] -%}
+            {%- if message['content'] is mapping -%}
+                {%- if 'name' in message['content'] and 'response' in message['content'] -%}
+                    {{ '<start_function_response>response:' + message['content']['name'] | trim + '{' }}
+                    {%- set response_ns = namespace(found_first=false) -%}
+                    {%- for key, value in message['content']['response'] | dictsort -%}
+                        {%- if response_ns.found_first %},{% endif -%}
+                        {%- set response_ns.found_first = true -%}
+                        {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                    {%- endfor -%}
+                    {{- '}<end_function_response>' -}}
+                {%- elif 'name' in message -%}
+                    {{ '<start_function_response>response:' + message['name'] | trim + '{' }}
+                    {%- set response_ns = namespace(found_first=false) -%}
+                    {%- for key, value in message['content'] | dictsort -%}
+                        {%- if response_ns.found_first %},{% endif -%}
+                        {%- set response_ns.found_first = true -%}
+                        {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                    {%- endfor -%}
+                    {{- '}<end_function_response>' -}}
+                {%- else -%}
+                    {{ raise_exception("Invalid tool response mapping: must contain 'name' and 'response' keys, or 'name' must be in the message.") }}
+                {%- endif -%}
+            {%- elif message['content'] is string -%}
+                {%- if 'name' in message -%}
+                     {{ '<start_function_response>response:' + message['name'] | trim + '{value:' + format_argument(message['content'], escape_keys=False) + '}<end_function_response>' }}
+                {%- else -%}
+                     {{ raise_exception("Invalid tool response: 'name' must be provided.") }}
+                {%- endif -%}
+            {%- elif message['content'] is sequence -%}
+                {%- for item in message['content'] -%}
+                    {%- if item is mapping -%}
+                        {%- if 'name' in item and 'response' in item -%}
+                            {{ '<start_function_response>response:' + item['name'] | trim + '{' }}
+                            {%- set response_ns = namespace(found_first=false) -%}
+                            {%- for key, value in item['response'] | dictsort -%}
+                                {%- if response_ns.found_first %},{% endif -%}
+                                {%- set response_ns.found_first = true -%}
+                                {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                            {%- endfor -%}
+                            {{- '}<end_function_response>' -}}
+                        {%- elif 'name' in message -%}
+                            {{ '<start_function_response>response:' + message['name'] | trim + '{' }}
+                            {%- set response_ns = namespace(found_first=false) -%}
+                            {%- for key, value in item | dictsort -%}
+                                {%- if response_ns.found_first %},{% endif -%}
+                                {%- set response_ns.found_first = true -%}
+                                {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                            {%- endfor -%}
+                            {{- '}<end_function_response>' -}}
+                        {%- else -%}
+                            {{ raise_exception("Invalid tool response mapping: must contain 'name' and 'response' keys, or 'name' must be in the message.") }}
+                        {%- endif -%}
+                    {%- else -%}
+                        {{ raise_exception("Invalid tool response message: multiple responses must all be mappings") }}
+                    {%- endif -%}
+                {%- endfor -%}
+            {%- else -%}
+                {{ raise_exception("Invalid content type in tool message: must be mapping, sequence of mappings, or string.") }}
+            {%- endif -%}
+        {%- endif -%}
+        {%- set ns.prev_message_type = 'tool_response' -%}
+    {%- endif -%}
+    {%- if ns.prev_message_type not in ['tool_call', 'tool_response'] -%}
+        {{ '<end_of_turn>\n' }}
+    {%- endif -%}
+{%- endfor -%}
+{%- if add_generation_prompt -%}
+    {%- if ns.prev_message_type != 'tool_response' -%}
+        {{- '<start_of_turn>model\n' -}}
+    {%- endif -%}
+{%- endif -%}

checkpoint-154/config.json ADDED Viewed

	@@ -0,0 +1,62 @@

+{
+  "_sliding_window_pattern": 6,
+  "architectures": [
+    "Gemma3ForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "attn_logit_softcapping": null,
+  "bos_token_id": 2,
+  "dtype": "bfloat16",
+  "eos_token_id": 1,
+  "final_logit_softcapping": null,
+  "head_dim": 256,
+  "hidden_activation": "gelu_pytorch_tanh",
+  "hidden_size": 640,
+  "initializer_range": 0.02,
+  "intermediate_size": 2048,
+  "layer_types": [
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention"
+  ],
+  "max_position_embeddings": 32768,
+  "model_type": "gemma3_text",
+  "num_attention_heads": 4,
+  "num_hidden_layers": 18,
+  "num_key_value_heads": 1,
+  "pad_token_id": 0,
+  "query_pre_attn_scalar": 256,
+  "rms_norm_eps": 1e-06,
+  "rope_parameters": {
+    "full_attention": {
+      "rope_theta": 1000000.0,
+      "rope_type": "default"
+    },
+    "sliding_attention": {
+      "rope_theta": 10000.0,
+      "rope_type": "default"
+    }
+  },
+  "sliding_window": 512,
+  "tie_word_embeddings": true,
+  "transformers_version": "5.5.1",
+  "use_bidirectional_attention": false,
+  "use_cache": false,
+  "vocab_size": 262144
+}

checkpoint-154/generation_config.json ADDED Viewed

	@@ -0,0 +1,15 @@

+{
+  "bos_token_id": 2,
+  "cache_implementation": "hybrid",
+  "do_sample": true,
+  "eos_token_id": [
+    1,
+    1,
+    50,
+    106
+  ],
+  "pad_token_id": 0,
+  "top_k": 64,
+  "top_p": 0.95,
+  "transformers_version": "5.5.1"
+}

checkpoint-154/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:93fb534a4891753c3ebb1a9de98b341c468d2af48358952ad356dba5d9a6c73a
+size 536223056

checkpoint-154/optimizer.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c6462a18da18cd52b1d0a0207bad998d42dc86e0f7df8492a865e5145bb25c45
+size 1072594443

checkpoint-154/rng_state.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:718a0f3db00824213036a2c0441849791319b7d9cf189065873bb26a7020738e
+size 14645

checkpoint-154/scheduler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:d1e83b8ffb0782b40f36dd8317e0757ffe7f134c174b4c60d0bd74dcd9d506e7
+size 1465

checkpoint-154/tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:80d7f800b949accd7eb940bac75e642f9468e4df157403032a55bf54ed23b650
+size 33384898

checkpoint-154/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,26 @@

+{
+  "backend": "tokenizers",
+  "boi_token": "<start_of_image>",
+  "bos_token": "<bos>",
+  "clean_up_tokenization_spaces": false,
+  "eoi_token": "<end_of_image>",
+  "eos_token": "<eos>",
+  "image_token": "<image_soft_token>",
+  "is_local": false,
+  "mask_token": "<mask>",
+  "model_max_length": 1000000000000000019884624838656,
+  "model_specific_special_tokens": {
+    "boi_token": "<start_of_image>",
+    "eoi_token": "<end_of_image>",
+    "image_token": "<image_soft_token>",
+    "sfr_token": "<start_function_response>"
+  },
+  "pad_token": "<pad>",
+  "padding_side": "left",
+  "sfr_token": "<start_function_response>",
+  "sp_model_kwargs": null,
+  "spaces_between_special_tokens": false,
+  "tokenizer_class": "GemmaTokenizer",
+  "unk_token": "<unk>",
+  "use_default_system_prompt": false
+}

checkpoint-154/trainer_state.json ADDED Viewed

	@@ -0,0 +1,97 @@

+{
+  "best_global_step": null,
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 2.0,
+  "eval_steps": 50,
+  "global_step": 154,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "entropy": 0.8290567851066589,
+      "epoch": 0.6568144499178982,
+      "grad_norm": 0.75390625,
+      "learning_rate": 9.388394947836278e-06,
+      "loss": 0.11062156677246093,
+      "mean_token_accuracy": 0.9778690934181213,
+      "num_tokens": 389732.0,
+      "step": 50
+    },
+    {
+      "epoch": 0.6568144499178982,
+      "eval_entropy": 0.7784549756483599,
+      "eval_loss": 0.03614399954676628,
+      "eval_mean_token_accuracy": 0.9917981918756064,
+      "eval_num_tokens": 389732.0,
+      "eval_runtime": 12.9713,
+      "eval_samples_per_second": 23.513,
+      "eval_steps_per_second": 5.936,
+      "step": 50
+    },
+    {
+      "entropy": 0.7608775209834557,
+      "epoch": 1.3021346469622332,
+      "grad_norm": 0.6796875,
+      "learning_rate": 7.660160382576683e-06,
+      "loss": 0.022044627666473388,
+      "mean_token_accuracy": 0.9940849557784374,
+      "num_tokens": 771178.0,
+      "step": 100
+    },
+    {
+      "epoch": 1.3021346469622332,
+      "eval_entropy": 0.7180899474527929,
+      "eval_loss": 0.019734159111976624,
+      "eval_mean_token_accuracy": 0.9949120081864394,
+      "eval_num_tokens": 771178.0,
+      "eval_runtime": 13.0354,
+      "eval_samples_per_second": 23.398,
+      "eval_steps_per_second": 5.907,
+      "step": 100
+    },
+    {
+      "entropy": 0.7120695394277573,
+      "epoch": 1.9589490968801315,
+      "grad_norm": 0.4453125,
+      "learning_rate": 5.25488887635095e-06,
+      "loss": 0.013736556768417358,
+      "mean_token_accuracy": 0.9967586588859558,
+      "num_tokens": 1165084.0,
+      "step": 150
+    },
+    {
+      "epoch": 1.9589490968801315,
+      "eval_entropy": 0.6842893903905695,
+      "eval_loss": 0.013124481774866581,
+      "eval_mean_token_accuracy": 0.9965388542645938,
+      "eval_num_tokens": 1165084.0,
+      "eval_runtime": 13.0811,
+      "eval_samples_per_second": 23.316,
+      "eval_steps_per_second": 5.886,
+      "step": 150
+    }
+  ],
+  "logging_steps": 50,
+  "max_steps": 308,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 4,
+  "save_steps": 500,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": false
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 1006601818940928.0,
+  "train_batch_size": 4,
+  "trial_name": null,
+  "trial_params": null
+}

checkpoint-154/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e5363ce9d99dd9dbf1bc2c5ff6c4b0ce553147005c7381cbde8798f93d5971fb
+size 5649

checkpoint-231/chat_template.jinja ADDED Viewed

	@@ -0,0 +1,279 @@

+{%- macro format_parameters(properties, required) -%}
+    {%- set standard_keys = ['description', 'type', 'properties', 'required', 'nullable'] -%}
+    {%- set ns = namespace(found_first=false) -%}
+    {%- for key, value in properties | dictsort -%}
+        {%- if key not in standard_keys -%}
+            {%- if ns.found_first %},{% endif -%}
+            {%- set ns.found_first = true -%}
+            {{- key }}:{description:<escape>{{ value['description'] }}<escape>
+            {%- if value['type'] | upper == 'STRING' -%}
+                {%- if value['enum'] -%}
+                    ,enum:{{ format_argument(value['enum']) }}
+                {%- endif -%}
+            {%- elif value['type'] | upper == 'OBJECT' -%}
+                ,properties:{
+                {%- if value['properties'] is defined and value['properties'] is mapping -%}
+                    {{- format_parameters(value['properties'], value['required'] | default([])) -}}
+                {%- elif value is mapping -%}
+                    {{- format_parameters(value, value['required'] | default([])) -}}
+                {%- endif -%}
+                }
+                {%- if value['required'] -%}
+                    ,required:[
+                    {%- for item in value['required'] | default([]) -%}
+                        <escape>{{- item -}}<escape>
+                        {%- if not loop.last %},{% endif -%}
+                    {%- endfor -%}
+                    ]
+                {%- endif -%}
+            {%- elif value['type'] | upper == 'ARRAY' -%}
+                {%- if value['items'] is mapping and value['items'] -%}
+                    ,items:{
+                    {%- set ns_items = namespace(found_first=false) -%}
+                    {%- for item_key, item_value in value['items'] | dictsort -%}
+                        {%- if item_value is not none -%}
+                            {%- if ns_items.found_first %},{% endif -%}
+                            {%- set ns_items.found_first = true -%}
+                            {%- if item_key == 'properties' -%}
+                                properties:{
+                                {%- if item_value is mapping -%}
+                                    {{- format_parameters(item_value, value['items']['required'] | default([])) -}}
+                                {%- endif -%}
+                                }
+                            {%- elif item_key == 'required' -%}
+                                required:[
+                                {%- for req_item in item_value -%}
+                                    <escape>{{- req_item -}}<escape>
+                                    {%- if not loop.last %},{% endif -%}
+                                {%- endfor -%}
+                                ]
+                            {%- elif item_key == 'type' -%}
+                                {%- if item_value is string -%}
+                                    type:{{ format_argument(item_value | upper) }}
+                                {%- else -%}
+                                    type:{{ format_argument(item_value | map('upper') | list) }}
+                                {%- endif -%}
+                            {%- else -%}
+                                {{ item_key }}:{{ format_argument(item_value) }}
+                            {%- endif -%}
+                        {%- endif -%}
+                    {%- endfor -%}
+                    }
+                {%- endif -%}
+            {%- endif -%}
+            ,type:<escape>{{ value['type'] | upper }}<escape>}
+        {%- endif -%}
+    {%- endfor -%}
+{%- endmacro -%}
+{% macro format_function_declaration(tool_data) -%}
+declaration:{{- tool_data['function']['name'] -}}
+{description:<escape>{{- tool_data['function']['description'] -}}<escape>
+{%- set params = tool_data['function']['parameters'] -%}
+{%- if params -%}
+    ,parameters:{
+    {%- if params['properties'] -%}
+        properties:{ {{- format_parameters(params['properties'], params['required']) -}} },
+    {%- endif -%}
+    {%- if params['required'] -%}
+        required:[
+        {%- for item in params['required'] -%}
+            <escape>{{- item -}}<escape>
+            {{- ',' if not loop.last -}}
+        {%- endfor -%}
+        ],
+    {%- endif -%}
+    {%- if params['type'] -%}
+        type:<escape>{{- params['type'] | upper -}}<escape>}
+    {%- endif -%}
+{%- endif -%}
+}
+{%- endmacro -%}
+{% macro format_argument(argument, escape_keys=True) -%}
+{%- if argument is string -%}
+    {{- '<escape>' + argument + '<escape>' -}}
+{%- elif argument is boolean -%}
+    {%- if argument -%}
+        {{- 'true' -}}
+    {%- else -%}
+        {{- 'false' -}}
+    {%- endif -%}
+{%- elif argument is mapping -%}
+    {{- '{' -}}
+    {%- set ns = namespace(found_first=false) -%}
+    {%- for key, value in argument | dictsort -%}
+        {%- if ns.found_first %},{% endif -%}
+        {%- set ns.found_first = true -%}
+        {%- if escape_keys -%}
+            {{- '<escape>' + key + '<escape>' -}}
+        {%- else -%}
+            {{- key -}}
+        {%- endif -%}
+        :{{- format_argument(value, escape_keys=escape_keys) -}}
+    {%- endfor -%}
+    {{- '}' -}}
+{%- elif argument is sequence -%}
+    {{- '[' -}}
+    {%- for item in argument -%}
+        {{- format_argument(item, escape_keys=escape_keys) -}}
+        {%- if not loop.last %},{% endif -%}
+    {%- endfor -%}
+    {{- ']' -}}
+{%- else -%}
+    {{- argument -}}
+{%- endif -%}
+{%- endmacro -%}
+{{ bos_token }}
+{%- set ns = namespace(prev_message_type=None) -%}
+{#- Tool Declarations -#}
+{%- set loop_messages = messages -%}
+{%- if tools or messages[0]['role'] == 'system' or messages[0]['role'] == 'developer' -%}
+    {{- '<start_of_turn>developer\n' -}}
+    {%- if messages[0]['role'] == 'system' or messages[0]['role'] == 'developer' -%}
+        {%- if messages[0]['content'] is string -%}
+            {{- messages[0]['content'] | trim -}}
+        {%- elif messages[0]['content'] is sequence -%}
+            {%- for item in messages[0]['content'] -%}
+                {%- if item['type'] == 'text' -%}
+                    {{- item['text'] | trim -}}
+                {%- endif -%}
+            {%- endfor -%}
+        {%- endif -%}
+        {%- set loop_messages = messages[1:] -%}
+    {%- endif -%}
+    {%- if tools -%}
+        {%- for tool in tools %}
+            {{- '<start_function_declaration>' -}}
+            {{- format_function_declaration(tool) | trim }}
+            {{- '<end_function_declaration>' -}}
+        {%- endfor %}
+    {%- endif -%}
+    {{- '<end_of_turn>\n' }}
+{%- endif %}
+{#- Loop through messages. -#}
+{%- for message in loop_messages -%}
+    {%- if (message['role'] == 'assistant') -%}
+        {#- Rename "assistant" to "model". -#}
+        {%- set role = "model" -%}
+    {%- else -%}
+        {%- set role = message['role'] -%}
+    {%- endif -%}
+    {%- if role != 'tool' -%}
+        {%- if ns.prev_message_type != 'tool_response' -%}
+            {{- '<start_of_turn>' + role + '\n' }}
+        {%- endif -%}
+        {%- set ns.prev_message_type = None -%}
+        {%- if 'content' in message and message['content'] is not none -%}
+            {%- if message['content'] is string -%}
+                {{ message['content'] | trim }}
+            {%- elif message['content'] is sequence -%}
+                {%- for item in message['content'] -%}
+                    {%- if item['type'] == 'image' -%}
+                        {{ '<start_of_image>' }}
+                    {%- elif item['type'] == 'text' -%}
+                        {{ item['text'] | trim }}
+                    {%- endif -%}
+                {%- endfor -%}
+            {%- else -%}
+                {{ raise_exception("Invalid content type in user/assistant message") }}
+            {%- endif -%}
+            {%- set ns.prev_message_type = 'content' -%}
+        {%- endif -%}
+        {%- if 'tool_calls' in message and message['tool_calls'] and message['tool_calls'] is iterable -%}
+            {#- Tool Calls -#}
+            {%- for tool_call in message['tool_calls'] -%}
+                {% set function = tool_call['function'] %}
+                {{-  '<start_function_call>call:' + function['name'] + '{' -}}
+                {%- if 'arguments' in function -%}
+                    {%- if function['arguments'] is mapping -%}
+                        {%- set ns = namespace(found_first=false) -%}
+                        {%- for key, value in function['arguments'] | dictsort -%}
+                            {%- if ns.found_first %},{% endif -%}
+                            {%- set ns.found_first = true -%}
+                            {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                        {%- endfor -%}
+                    {%- elif function['arguments'] is string -%}
+                        {# This handles string-JSON, just in case #}
+                    {{ function['arguments'] }}
+                    {%- endif %}
+                {%- endif -%}
+                {{- '}<end_function_call>' -}}
+            {%- endfor -%}
+            {%- if loop.last -%}
+                {{ '<start_function_response>' }}
+            {%- endif -%}
+            {%- set ns.prev_message_type = 'tool_call' -%}
+        {%- endif -%}
+    {%- else -%}
+        {#- Tool Responses -#}
+        {%- if 'content' in message and message['content'] -%}
+            {%- if message['content'] is mapping -%}
+                {%- if 'name' in message['content'] and 'response' in message['content'] -%}
+                    {{ '<start_function_response>response:' + message['content']['name'] | trim + '{' }}
+                    {%- set response_ns = namespace(found_first=false) -%}
+                    {%- for key, value in message['content']['response'] | dictsort -%}
+                        {%- if response_ns.found_first %},{% endif -%}
+                        {%- set response_ns.found_first = true -%}
+                        {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                    {%- endfor -%}
+                    {{- '}<end_function_response>' -}}
+                {%- elif 'name' in message -%}
+                    {{ '<start_function_response>response:' + message['name'] | trim + '{' }}
+                    {%- set response_ns = namespace(found_first=false) -%}
+                    {%- for key, value in message['content'] | dictsort -%}
+                        {%- if response_ns.found_first %},{% endif -%}
+                        {%- set response_ns.found_first = true -%}
+                        {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                    {%- endfor -%}
+                    {{- '}<end_function_response>' -}}
+                {%- else -%}
+                    {{ raise_exception("Invalid tool response mapping: must contain 'name' and 'response' keys, or 'name' must be in the message.") }}
+                {%- endif -%}
+            {%- elif message['content'] is string -%}
+                {%- if 'name' in message -%}
+                     {{ '<start_function_response>response:' + message['name'] | trim + '{value:' + format_argument(message['content'], escape_keys=False) + '}<end_function_response>' }}
+                {%- else -%}
+                     {{ raise_exception("Invalid tool response: 'name' must be provided.") }}
+                {%- endif -%}
+            {%- elif message['content'] is sequence -%}
+                {%- for item in message['content'] -%}
+                    {%- if item is mapping -%}
+                        {%- if 'name' in item and 'response' in item -%}
+                            {{ '<start_function_response>response:' + item['name'] | trim + '{' }}
+                            {%- set response_ns = namespace(found_first=false) -%}
+                            {%- for key, value in item['response'] | dictsort -%}
+                                {%- if response_ns.found_first %},{% endif -%}
+                                {%- set response_ns.found_first = true -%}
+                                {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                            {%- endfor -%}
+                            {{- '}<end_function_response>' -}}
+                        {%- elif 'name' in message -%}
+                            {{ '<start_function_response>response:' + message['name'] | trim + '{' }}
+                            {%- set response_ns = namespace(found_first=false) -%}
+                            {%- for key, value in item | dictsort -%}
+                                {%- if response_ns.found_first %},{% endif -%}
+                                {%- set response_ns.found_first = true -%}
+                                {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                            {%- endfor -%}
+                            {{- '}<end_function_response>' -}}
+                        {%- else -%}
+                            {{ raise_exception("Invalid tool response mapping: must contain 'name' and 'response' keys, or 'name' must be in the message.") }}
+                        {%- endif -%}
+                    {%- else -%}
+                        {{ raise_exception("Invalid tool response message: multiple responses must all be mappings") }}
+                    {%- endif -%}
+                {%- endfor -%}
+            {%- else -%}
+                {{ raise_exception("Invalid content type in tool message: must be mapping, sequence of mappings, or string.") }}
+            {%- endif -%}
+        {%- endif -%}
+        {%- set ns.prev_message_type = 'tool_response' -%}
+    {%- endif -%}
+    {%- if ns.prev_message_type not in ['tool_call', 'tool_response'] -%}
+        {{ '<end_of_turn>\n' }}
+    {%- endif -%}
+{%- endfor -%}
+{%- if add_generation_prompt -%}
+    {%- if ns.prev_message_type != 'tool_response' -%}
+        {{- '<start_of_turn>model\n' -}}
+    {%- endif -%}
+{%- endif -%}

checkpoint-231/config.json ADDED Viewed

	@@ -0,0 +1,62 @@

+{
+  "_sliding_window_pattern": 6,
+  "architectures": [
+    "Gemma3ForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "attn_logit_softcapping": null,
+  "bos_token_id": 2,
+  "dtype": "bfloat16",
+  "eos_token_id": 1,
+  "final_logit_softcapping": null,
+  "head_dim": 256,
+  "hidden_activation": "gelu_pytorch_tanh",
+  "hidden_size": 640,
+  "initializer_range": 0.02,
+  "intermediate_size": 2048,
+  "layer_types": [
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention"
+  ],
+  "max_position_embeddings": 32768,
+  "model_type": "gemma3_text",
+  "num_attention_heads": 4,
+  "num_hidden_layers": 18,
+  "num_key_value_heads": 1,
+  "pad_token_id": 0,
+  "query_pre_attn_scalar": 256,
+  "rms_norm_eps": 1e-06,
+  "rope_parameters": {
+    "full_attention": {
+      "rope_theta": 1000000.0,
+      "rope_type": "default"
+    },
+    "sliding_attention": {
+      "rope_theta": 10000.0,
+      "rope_type": "default"
+    }
+  },
+  "sliding_window": 512,
+  "tie_word_embeddings": true,
+  "transformers_version": "5.5.1",
+  "use_bidirectional_attention": false,
+  "use_cache": false,
+  "vocab_size": 262144
+}

checkpoint-231/generation_config.json ADDED Viewed

	@@ -0,0 +1,15 @@

+{
+  "bos_token_id": 2,
+  "cache_implementation": "hybrid",
+  "do_sample": true,
+  "eos_token_id": [
+    1,
+    1,
+    50,
+    106
+  ],
+  "pad_token_id": 0,
+  "top_k": 64,
+  "top_p": 0.95,
+  "transformers_version": "5.5.1"
+}

checkpoint-231/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:55b9632e5cf5540c78cd35b1e0e220c54e92aff80343623998b0b23885f9b141
+size 536223056

checkpoint-231/optimizer.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:6ea1cca6a4544a5b70c2fd53427f81667a461bda7b06d55017ebe5de993aa40c
+size 1072594443

checkpoint-231/rng_state.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f196323d7423b60f8e4ceb7dbf8715ee326c0d068e5ff164f13c63b279b9f1a0
+size 14645

checkpoint-231/scheduler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e2c2b23c7e4352465f050c1d63ce488c1582e84995535f53d01e1408547e53ea
+size 1465

checkpoint-231/tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:80d7f800b949accd7eb940bac75e642f9468e4df157403032a55bf54ed23b650
+size 33384898

checkpoint-231/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,26 @@

+{
+  "backend": "tokenizers",
+  "boi_token": "<start_of_image>",
+  "bos_token": "<bos>",
+  "clean_up_tokenization_spaces": false,
+  "eoi_token": "<end_of_image>",
+  "eos_token": "<eos>",
+  "image_token": "<image_soft_token>",
+  "is_local": false,
+  "mask_token": "<mask>",
+  "model_max_length": 1000000000000000019884624838656,
+  "model_specific_special_tokens": {
+    "boi_token": "<start_of_image>",
+    "eoi_token": "<end_of_image>",
+    "image_token": "<image_soft_token>",
+    "sfr_token": "<start_function_response>"
+  },
+  "pad_token": "<pad>",
+  "padding_side": "left",
+  "sfr_token": "<start_function_response>",
+  "sp_model_kwargs": null,
+  "spaces_between_special_tokens": false,
+  "tokenizer_class": "GemmaTokenizer",
+  "unk_token": "<unk>",
+  "use_default_system_prompt": false
+}

checkpoint-231/trainer_state.json ADDED Viewed

	@@ -0,0 +1,118 @@

+{
+  "best_global_step": null,
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 3.0,
+  "eval_steps": 50,
+  "global_step": 231,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "entropy": 0.8290567851066589,
+      "epoch": 0.6568144499178982,
+      "grad_norm": 0.75390625,
+      "learning_rate": 9.388394947836278e-06,
+      "loss": 0.11062156677246093,
+      "mean_token_accuracy": 0.9778690934181213,
+      "num_tokens": 389732.0,
+      "step": 50
+    },
+    {
+      "epoch": 0.6568144499178982,
+      "eval_entropy": 0.7784549756483599,
+      "eval_loss": 0.03614399954676628,
+      "eval_mean_token_accuracy": 0.9917981918756064,
+      "eval_num_tokens": 389732.0,
+      "eval_runtime": 12.9713,
+      "eval_samples_per_second": 23.513,
+      "eval_steps_per_second": 5.936,
+      "step": 50
+    },
+    {
+      "entropy": 0.7608775209834557,
+      "epoch": 1.3021346469622332,
+      "grad_norm": 0.6796875,
+      "learning_rate": 7.660160382576683e-06,
+      "loss": 0.022044627666473388,
+      "mean_token_accuracy": 0.9940849557784374,
+      "num_tokens": 771178.0,
+      "step": 100
+    },
+    {
+      "epoch": 1.3021346469622332,
+      "eval_entropy": 0.7180899474527929,
+      "eval_loss": 0.019734159111976624,
+      "eval_mean_token_accuracy": 0.9949120081864394,
+      "eval_num_tokens": 771178.0,
+      "eval_runtime": 13.0354,
+      "eval_samples_per_second": 23.398,
+      "eval_steps_per_second": 5.907,
+      "step": 100
+    },
+    {
+      "entropy": 0.7120695394277573,
+      "epoch": 1.9589490968801315,
+      "grad_norm": 0.4453125,
+      "learning_rate": 5.25488887635095e-06,
+      "loss": 0.013736556768417358,
+      "mean_token_accuracy": 0.9967586588859558,
+      "num_tokens": 1165084.0,
+      "step": 150
+    },
+    {
+      "epoch": 1.9589490968801315,
+      "eval_entropy": 0.6842893903905695,
+      "eval_loss": 0.013124481774866581,
+      "eval_mean_token_accuracy": 0.9965388542645938,
+      "eval_num_tokens": 1165084.0,
+      "eval_runtime": 13.0811,
+      "eval_samples_per_second": 23.316,
+      "eval_steps_per_second": 5.886,
+      "step": 150
+    },
+    {
+      "entropy": 0.6924438694961198,
+      "epoch": 2.6042692939244665,
+      "grad_norm": 0.423828125,
+      "learning_rate": 2.7847456480060476e-06,
+      "loss": 0.008616942763328552,
+      "mean_token_accuracy": 0.9980061287795011,
+      "num_tokens": 1550848.0,
+      "step": 200
+    },
+    {
+      "epoch": 2.6042692939244665,
+      "eval_entropy": 0.678199358961799,
+      "eval_loss": 0.012337171472609043,
+      "eval_mean_token_accuracy": 0.996950343831793,
+      "eval_num_tokens": 1550848.0,
+      "eval_runtime": 13.0386,
+      "eval_samples_per_second": 23.392,
+      "eval_steps_per_second": 5.906,
+      "step": 200
+    }
+  ],
+  "logging_steps": 50,
+  "max_steps": 308,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 4,
+  "save_steps": 500,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": false
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 1508776468555776.0,
+  "train_batch_size": 4,
+  "trial_name": null,
+  "trial_params": null
+}

checkpoint-231/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e5363ce9d99dd9dbf1bc2c5ff6c4b0ce553147005c7381cbde8798f93d5971fb
+size 5649

checkpoint-308/chat_template.jinja ADDED Viewed

	@@ -0,0 +1,279 @@

+{%- macro format_parameters(properties, required) -%}
+    {%- set standard_keys = ['description', 'type', 'properties', 'required', 'nullable'] -%}
+    {%- set ns = namespace(found_first=false) -%}
+    {%- for key, value in properties | dictsort -%}
+        {%- if key not in standard_keys -%}
+            {%- if ns.found_first %},{% endif -%}
+            {%- set ns.found_first = true -%}
+            {{- key }}:{description:<escape>{{ value['description'] }}<escape>
+            {%- if value['type'] | upper == 'STRING' -%}
+                {%- if value['enum'] -%}
+                    ,enum:{{ format_argument(value['enum']) }}
+                {%- endif -%}
+            {%- elif value['type'] | upper == 'OBJECT' -%}
+                ,properties:{
+                {%- if value['properties'] is defined and value['properties'] is mapping -%}
+                    {{- format_parameters(value['properties'], value['required'] | default([])) -}}
+                {%- elif value is mapping -%}
+                    {{- format_parameters(value, value['required'] | default([])) -}}
+                {%- endif -%}
+                }
+                {%- if value['required'] -%}
+                    ,required:[
+                    {%- for item in value['required'] | default([]) -%}
+                        <escape>{{- item -}}<escape>
+                        {%- if not loop.last %},{% endif -%}
+                    {%- endfor -%}
+                    ]
+                {%- endif -%}
+            {%- elif value['type'] | upper == 'ARRAY' -%}
+                {%- if value['items'] is mapping and value['items'] -%}
+                    ,items:{
+                    {%- set ns_items = namespace(found_first=false) -%}
+                    {%- for item_key, item_value in value['items'] | dictsort -%}
+                        {%- if item_value is not none -%}
+                            {%- if ns_items.found_first %},{% endif -%}
+                            {%- set ns_items.found_first = true -%}
+                            {%- if item_key == 'properties' -%}
+                                properties:{
+                                {%- if item_value is mapping -%}
+                                    {{- format_parameters(item_value, value['items']['required'] | default([])) -}}
+                                {%- endif -%}
+                                }
+                            {%- elif item_key == 'required' -%}
+                                required:[
+                                {%- for req_item in item_value -%}
+                                    <escape>{{- req_item -}}<escape>
+                                    {%- if not loop.last %},{% endif -%}
+                                {%- endfor -%}
+                                ]
+                            {%- elif item_key == 'type' -%}
+                                {%- if item_value is string -%}
+                                    type:{{ format_argument(item_value | upper) }}
+                                {%- else -%}
+                                    type:{{ format_argument(item_value | map('upper') | list) }}
+                                {%- endif -%}
+                            {%- else -%}
+                                {{ item_key }}:{{ format_argument(item_value) }}
+                            {%- endif -%}
+                        {%- endif -%}
+                    {%- endfor -%}
+                    }
+                {%- endif -%}
+            {%- endif -%}
+            ,type:<escape>{{ value['type'] | upper }}<escape>}
+        {%- endif -%}
+    {%- endfor -%}
+{%- endmacro -%}
+{% macro format_function_declaration(tool_data) -%}
+declaration:{{- tool_data['function']['name'] -}}
+{description:<escape>{{- tool_data['function']['description'] -}}<escape>
+{%- set params = tool_data['function']['parameters'] -%}
+{%- if params -%}
+    ,parameters:{
+    {%- if params['properties'] -%}
+        properties:{ {{- format_parameters(params['properties'], params['required']) -}} },
+    {%- endif -%}
+    {%- if params['required'] -%}
+        required:[
+        {%- for item in params['required'] -%}
+            <escape>{{- item -}}<escape>
+            {{- ',' if not loop.last -}}
+        {%- endfor -%}
+        ],
+    {%- endif -%}
+    {%- if params['type'] -%}
+        type:<escape>{{- params['type'] | upper -}}<escape>}
+    {%- endif -%}
+{%- endif -%}
+}
+{%- endmacro -%}
+{% macro format_argument(argument, escape_keys=True) -%}
+{%- if argument is string -%}
+    {{- '<escape>' + argument + '<escape>' -}}
+{%- elif argument is boolean -%}
+    {%- if argument -%}
+        {{- 'true' -}}
+    {%- else -%}
+        {{- 'false' -}}
+    {%- endif -%}
+{%- elif argument is mapping -%}
+    {{- '{' -}}
+    {%- set ns = namespace(found_first=false) -%}
+    {%- for key, value in argument | dictsort -%}
+        {%- if ns.found_first %},{% endif -%}
+        {%- set ns.found_first = true -%}
+        {%- if escape_keys -%}
+            {{- '<escape>' + key + '<escape>' -}}
+        {%- else -%}
+            {{- key -}}
+        {%- endif -%}
+        :{{- format_argument(value, escape_keys=escape_keys) -}}
+    {%- endfor -%}
+    {{- '}' -}}
+{%- elif argument is sequence -%}
+    {{- '[' -}}
+    {%- for item in argument -%}
+        {{- format_argument(item, escape_keys=escape_keys) -}}
+        {%- if not loop.last %},{% endif -%}
+    {%- endfor -%}
+    {{- ']' -}}
+{%- else -%}
+    {{- argument -}}
+{%- endif -%}
+{%- endmacro -%}
+{{ bos_token }}
+{%- set ns = namespace(prev_message_type=None) -%}
+{#- Tool Declarations -#}
+{%- set loop_messages = messages -%}
+{%- if tools or messages[0]['role'] == 'system' or messages[0]['role'] == 'developer' -%}
+    {{- '<start_of_turn>developer\n' -}}
+    {%- if messages[0]['role'] == 'system' or messages[0]['role'] == 'developer' -%}
+        {%- if messages[0]['content'] is string -%}
+            {{- messages[0]['content'] | trim -}}
+        {%- elif messages[0]['content'] is sequence -%}
+            {%- for item in messages[0]['content'] -%}
+                {%- if item['type'] == 'text' -%}
+                    {{- item['text'] | trim -}}
+                {%- endif -%}
+            {%- endfor -%}
+        {%- endif -%}
+        {%- set loop_messages = messages[1:] -%}
+    {%- endif -%}
+    {%- if tools -%}
+        {%- for tool in tools %}
+            {{- '<start_function_declaration>' -}}
+            {{- format_function_declaration(tool) | trim }}
+            {{- '<end_function_declaration>' -}}
+        {%- endfor %}
+    {%- endif -%}
+    {{- '<end_of_turn>\n' }}
+{%- endif %}
+{#- Loop through messages. -#}
+{%- for message in loop_messages -%}
+    {%- if (message['role'] == 'assistant') -%}
+        {#- Rename "assistant" to "model". -#}
+        {%- set role = "model" -%}
+    {%- else -%}
+        {%- set role = message['role'] -%}
+    {%- endif -%}
+    {%- if role != 'tool' -%}
+        {%- if ns.prev_message_type != 'tool_response' -%}
+            {{- '<start_of_turn>' + role + '\n' }}
+        {%- endif -%}
+        {%- set ns.prev_message_type = None -%}
+        {%- if 'content' in message and message['content'] is not none -%}
+            {%- if message['content'] is string -%}
+                {{ message['content'] | trim }}
+            {%- elif message['content'] is sequence -%}
+                {%- for item in message['content'] -%}
+                    {%- if item['type'] == 'image' -%}
+                        {{ '<start_of_image>' }}
+                    {%- elif item['type'] == 'text' -%}
+                        {{ item['text'] | trim }}
+                    {%- endif -%}
+                {%- endfor -%}
+            {%- else -%}
+                {{ raise_exception("Invalid content type in user/assistant message") }}
+            {%- endif -%}
+            {%- set ns.prev_message_type = 'content' -%}
+        {%- endif -%}
+        {%- if 'tool_calls' in message and message['tool_calls'] and message['tool_calls'] is iterable -%}
+            {#- Tool Calls -#}
+            {%- for tool_call in message['tool_calls'] -%}
+                {% set function = tool_call['function'] %}
+                {{-  '<start_function_call>call:' + function['name'] + '{' -}}
+                {%- if 'arguments' in function -%}
+                    {%- if function['arguments'] is mapping -%}
+                        {%- set ns = namespace(found_first=false) -%}
+                        {%- for key, value in function['arguments'] | dictsort -%}
+                            {%- if ns.found_first %},{% endif -%}
+                            {%- set ns.found_first = true -%}
+                            {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                        {%- endfor -%}
+                    {%- elif function['arguments'] is string -%}
+                        {# This handles string-JSON, just in case #}
+                    {{ function['arguments'] }}
+                    {%- endif %}
+                {%- endif -%}
+                {{- '}<end_function_call>' -}}
+            {%- endfor -%}
+            {%- if loop.last -%}
+                {{ '<start_function_response>' }}
+            {%- endif -%}
+            {%- set ns.prev_message_type = 'tool_call' -%}
+        {%- endif -%}
+    {%- else -%}
+        {#- Tool Responses -#}
+        {%- if 'content' in message and message['content'] -%}
+            {%- if message['content'] is mapping -%}
+                {%- if 'name' in message['content'] and 'response' in message['content'] -%}
+                    {{ '<start_function_response>response:' + message['content']['name'] | trim + '{' }}
+                    {%- set response_ns = namespace(found_first=false) -%}
+                    {%- for key, value in message['content']['response'] | dictsort -%}
+                        {%- if response_ns.found_first %},{% endif -%}
+                        {%- set response_ns.found_first = true -%}
+                        {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                    {%- endfor -%}
+                    {{- '}<end_function_response>' -}}
+                {%- elif 'name' in message -%}
+                    {{ '<start_function_response>response:' + message['name'] | trim + '{' }}
+                    {%- set response_ns = namespace(found_first=false) -%}
+                    {%- for key, value in message['content'] | dictsort -%}
+                        {%- if response_ns.found_first %},{% endif -%}
+                        {%- set response_ns.found_first = true -%}
+                        {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                    {%- endfor -%}
+                    {{- '}<end_function_response>' -}}
+                {%- else -%}
+                    {{ raise_exception("Invalid tool response mapping: must contain 'name' and 'response' keys, or 'name' must be in the message.") }}
+                {%- endif -%}
+            {%- elif message['content'] is string -%}
+                {%- if 'name' in message -%}
+                     {{ '<start_function_response>response:' + message['name'] | trim + '{value:' + format_argument(message['content'], escape_keys=False) + '}<end_function_response>' }}
+                {%- else -%}
+                     {{ raise_exception("Invalid tool response: 'name' must be provided.") }}
+                {%- endif -%}
+            {%- elif message['content'] is sequence -%}
+                {%- for item in message['content'] -%}
+                    {%- if item is mapping -%}
+                        {%- if 'name' in item and 'response' in item -%}
+                            {{ '<start_function_response>response:' + item['name'] | trim + '{' }}
+                            {%- set response_ns = namespace(found_first=false) -%}
+                            {%- for key, value in item['response'] | dictsort -%}
+                                {%- if response_ns.found_first %},{% endif -%}
+                                {%- set response_ns.found_first = true -%}
+                                {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                            {%- endfor -%}
+                            {{- '}<end_function_response>' -}}
+                        {%- elif 'name' in message -%}
+                            {{ '<start_function_response>response:' + message['name'] | trim + '{' }}
+                            {%- set response_ns = namespace(found_first=false) -%}
+                            {%- for key, value in item | dictsort -%}
+                                {%- if response_ns.found_first %},{% endif -%}
+                                {%- set response_ns.found_first = true -%}
+                                {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                            {%- endfor -%}
+                            {{- '}<end_function_response>' -}}
+                        {%- else -%}
+                            {{ raise_exception("Invalid tool response mapping: must contain 'name' and 'response' keys, or 'name' must be in the message.") }}
+                        {%- endif -%}
+                    {%- else -%}
+                        {{ raise_exception("Invalid tool response message: multiple responses must all be mappings") }}
+                    {%- endif -%}
+                {%- endfor -%}
+            {%- else -%}
+                {{ raise_exception("Invalid content type in tool message: must be mapping, sequence of mappings, or string.") }}
+            {%- endif -%}
+        {%- endif -%}
+        {%- set ns.prev_message_type = 'tool_response' -%}
+    {%- endif -%}
+    {%- if ns.prev_message_type not in ['tool_call', 'tool_response'] -%}
+        {{ '<end_of_turn>\n' }}
+    {%- endif -%}
+{%- endfor -%}
+{%- if add_generation_prompt -%}
+    {%- if ns.prev_message_type != 'tool_response' -%}
+        {{- '<start_of_turn>model\n' -}}
+    {%- endif -%}
+{%- endif -%}

checkpoint-308/config.json ADDED Viewed

	@@ -0,0 +1,62 @@

+{
+  "_sliding_window_pattern": 6,
+  "architectures": [
+    "Gemma3ForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "attn_logit_softcapping": null,
+  "bos_token_id": 2,
+  "dtype": "bfloat16",
+  "eos_token_id": 1,
+  "final_logit_softcapping": null,
+  "head_dim": 256,
+  "hidden_activation": "gelu_pytorch_tanh",
+  "hidden_size": 640,
+  "initializer_range": 0.02,
+  "intermediate_size": 2048,
+  "layer_types": [
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention"
+  ],
+  "max_position_embeddings": 32768,
+  "model_type": "gemma3_text",
+  "num_attention_heads": 4,
+  "num_hidden_layers": 18,
+  "num_key_value_heads": 1,
+  "pad_token_id": 0,
+  "query_pre_attn_scalar": 256,
+  "rms_norm_eps": 1e-06,
+  "rope_parameters": {
+    "full_attention": {
+      "rope_theta": 1000000.0,
+      "rope_type": "default"
+    },
+    "sliding_attention": {
+      "rope_theta": 10000.0,
+      "rope_type": "default"
+    }
+  },
+  "sliding_window": 512,
+  "tie_word_embeddings": true,
+  "transformers_version": "5.5.1",
+  "use_bidirectional_attention": false,
+  "use_cache": false,
+  "vocab_size": 262144
+}

checkpoint-308/generation_config.json ADDED Viewed

	@@ -0,0 +1,15 @@

+{
+  "bos_token_id": 2,
+  "cache_implementation": "hybrid",
+  "do_sample": true,
+  "eos_token_id": [
+    1,
+    1,
+    50,
+    106
+  ],
+  "pad_token_id": 0,
+  "top_k": 64,
+  "top_p": 0.95,
+  "transformers_version": "5.5.1"
+}

checkpoint-308/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:5fdad25a78c297aab9ec2b809949a6f0b1968f7a6a09d486343d48fe3f5c51da
+size 536223056

checkpoint-308/optimizer.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:251ca04ead3677dedb4846271e2205f3277a2eca493a068019cb64e3e4754342
+size 1072594443

checkpoint-308/rng_state.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:ea11996454b5587fcf33ae0ab5cf14b2031bf5f53f8c2ed5a48e87de31e29c84
+size 14645

checkpoint-308/scheduler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:46bf8c059ac0682006a9531cce7258bd0ef62fc0ab3b4eceb1892efacaf6680b
+size 1465

checkpoint-308/tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:80d7f800b949accd7eb940bac75e642f9468e4df157403032a55bf54ed23b650
+size 33384898

checkpoint-308/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,26 @@

+{
+  "backend": "tokenizers",
+  "boi_token": "<start_of_image>",
+  "bos_token": "<bos>",
+  "clean_up_tokenization_spaces": false,
+  "eoi_token": "<end_of_image>",
+  "eos_token": "<eos>",
+  "image_token": "<image_soft_token>",
+  "is_local": false,
+  "mask_token": "<mask>",
+  "model_max_length": 1000000000000000019884624838656,
+  "model_specific_special_tokens": {
+    "boi_token": "<start_of_image>",
+    "eoi_token": "<end_of_image>",
+    "image_token": "<image_soft_token>",
+    "sfr_token": "<start_function_response>"
+  },
+  "pad_token": "<pad>",
+  "padding_side": "left",
+  "sfr_token": "<start_function_response>",
+  "sp_model_kwargs": null,
+  "spaces_between_special_tokens": false,
+  "tokenizer_class": "GemmaTokenizer",
+  "unk_token": "<unk>",
+  "use_default_system_prompt": false
+}

checkpoint-308/trainer_state.json ADDED Viewed

	@@ -0,0 +1,171 @@

+{
+  "best_global_step": null,
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 4.0,
+  "eval_steps": 50,
+  "global_step": 308,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "entropy": 0.8290567851066589,
+      "epoch": 0.6568144499178982,
+      "grad_norm": 0.75390625,
+      "learning_rate": 9.388394947836278e-06,
+      "loss": 0.11062156677246093,
+      "mean_token_accuracy": 0.9778690934181213,
+      "num_tokens": 389732.0,
+      "step": 50
+    },
+    {
+      "epoch": 0.6568144499178982,
+      "eval_entropy": 0.7784549756483599,
+      "eval_loss": 0.03614399954676628,
+      "eval_mean_token_accuracy": 0.9917981918756064,
+      "eval_num_tokens": 389732.0,
+      "eval_runtime": 12.9713,
+      "eval_samples_per_second": 23.513,
+      "eval_steps_per_second": 5.936,
+      "step": 50
+    },
+    {
+      "entropy": 0.7608775209834557,
+      "epoch": 1.3021346469622332,
+      "grad_norm": 0.6796875,
+      "learning_rate": 7.660160382576683e-06,
+      "loss": 0.022044627666473388,
+      "mean_token_accuracy": 0.9940849557784374,
+      "num_tokens": 771178.0,
+      "step": 100
+    },
+    {
+      "epoch": 1.3021346469622332,
+      "eval_entropy": 0.7180899474527929,
+      "eval_loss": 0.019734159111976624,
+      "eval_mean_token_accuracy": 0.9949120081864394,
+      "eval_num_tokens": 771178.0,
+      "eval_runtime": 13.0354,
+      "eval_samples_per_second": 23.398,
+      "eval_steps_per_second": 5.907,
+      "step": 100
+    },
+    {
+      "entropy": 0.7120695394277573,
+      "epoch": 1.9589490968801315,
+      "grad_norm": 0.4453125,
+      "learning_rate": 5.25488887635095e-06,
+      "loss": 0.013736556768417358,
+      "mean_token_accuracy": 0.9967586588859558,
+      "num_tokens": 1165084.0,
+      "step": 150
+    },
+    {
+      "epoch": 1.9589490968801315,
+      "eval_entropy": 0.6842893903905695,
+      "eval_loss": 0.013124481774866581,
+      "eval_mean_token_accuracy": 0.9965388542645938,
+      "eval_num_tokens": 1165084.0,
+      "eval_runtime": 13.0811,
+      "eval_samples_per_second": 23.316,
+      "eval_steps_per_second": 5.886,
+      "step": 150
+    },
+    {
+      "entropy": 0.6924438694961198,
+      "epoch": 2.6042692939244665,
+      "grad_norm": 0.423828125,
+      "learning_rate": 2.7847456480060476e-06,
+      "loss": 0.008616942763328552,
+      "mean_token_accuracy": 0.9980061287795011,
+      "num_tokens": 1550848.0,
+      "step": 200
+    },
+    {
+      "epoch": 2.6042692939244665,
+      "eval_entropy": 0.678199358961799,
+      "eval_loss": 0.012337171472609043,
+      "eval_mean_token_accuracy": 0.996950343831793,
+      "eval_num_tokens": 1550848.0,
+      "eval_runtime": 13.0386,
+      "eval_samples_per_second": 23.392,
+      "eval_steps_per_second": 5.906,
+      "step": 200
+    },
+    {
+      "entropy": 0.6876007438312657,
+      "epoch": 3.2495894909688015,
+      "grad_norm": 0.6484375,
+      "learning_rate": 8.784064067287057e-07,
+      "loss": 0.008523799180984497,
+      "mean_token_accuracy": 0.9980273214915326,
+      "num_tokens": 1935501.0,
+      "step": 250
+    },
+    {
+      "epoch": 3.2495894909688015,
+      "eval_entropy": 0.6709755380432327,
+      "eval_loss": 0.011750386096537113,
+      "eval_mean_token_accuracy": 0.9970997378423616,
+      "eval_num_tokens": 1935501.0,
+      "eval_runtime": 13.0005,
+      "eval_samples_per_second": 23.461,
+      "eval_steps_per_second": 5.923,
+      "step": 250
+    },
+    {
+      "entropy": 0.6889240379631519,
+      "epoch": 3.9064039408866993,
+      "grad_norm": 0.6015625,
+      "learning_rate": 2.1053210266875346e-08,
+      "loss": 0.00826053500175476,
+      "mean_token_accuracy": 0.9980920545756817,
+      "num_tokens": 2322603.0,
+      "step": 300
+    },
+    {
+      "epoch": 3.9064039408866993,
+      "eval_entropy": 0.6722380197667456,
+      "eval_loss": 0.011939619667828083,
+      "eval_mean_token_accuracy": 0.996877890128594,
+      "eval_num_tokens": 2322603.0,
+      "eval_runtime": 13.2474,
+      "eval_samples_per_second": 23.023,
+      "eval_steps_per_second": 5.812,
+      "step": 300
+    },
+    {
+      "epoch": 4.0,
+      "eval_entropy": 0.6724205961475125,
+      "eval_loss": 0.01190057210624218,
+      "eval_mean_token_accuracy": 0.9969809620411365,
+      "eval_num_tokens": 2378596.0,
+      "eval_runtime": 13.6271,
+      "eval_samples_per_second": 22.382,
+      "eval_steps_per_second": 5.651,
+      "step": 308
+    }
+  ],
+  "logging_steps": 50,
+  "max_steps": 308,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 4,
+  "save_steps": 500,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": true
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 2009014625409792.0,
+  "train_batch_size": 4,
+  "trial_name": null,
+  "trial_params": null
+}

checkpoint-308/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e5363ce9d99dd9dbf1bc2c5ff6c4b0ce553147005c7381cbde8798f93d5971fb
+size 5649

checkpoint-77/chat_template.jinja ADDED Viewed

	@@ -0,0 +1,279 @@

+{%- macro format_parameters(properties, required) -%}
+    {%- set standard_keys = ['description', 'type', 'properties', 'required', 'nullable'] -%}
+    {%- set ns = namespace(found_first=false) -%}
+    {%- for key, value in properties | dictsort -%}
+        {%- if key not in standard_keys -%}
+            {%- if ns.found_first %},{% endif -%}
+            {%- set ns.found_first = true -%}
+            {{- key }}:{description:<escape>{{ value['description'] }}<escape>
+            {%- if value['type'] | upper == 'STRING' -%}
+                {%- if value['enum'] -%}
+                    ,enum:{{ format_argument(value['enum']) }}
+                {%- endif -%}
+            {%- elif value['type'] | upper == 'OBJECT' -%}
+                ,properties:{
+                {%- if value['properties'] is defined and value['properties'] is mapping -%}
+                    {{- format_parameters(value['properties'], value['required'] | default([])) -}}
+                {%- elif value is mapping -%}
+                    {{- format_parameters(value, value['required'] | default([])) -}}
+                {%- endif -%}
+                }
+                {%- if value['required'] -%}
+                    ,required:[
+                    {%- for item in value['required'] | default([]) -%}
+                        <escape>{{- item -}}<escape>
+                        {%- if not loop.last %},{% endif -%}
+                    {%- endfor -%}
+                    ]
+                {%- endif -%}
+            {%- elif value['type'] | upper == 'ARRAY' -%}
+                {%- if value['items'] is mapping and value['items'] -%}
+                    ,items:{
+                    {%- set ns_items = namespace(found_first=false) -%}
+                    {%- for item_key, item_value in value['items'] | dictsort -%}
+                        {%- if item_value is not none -%}
+                            {%- if ns_items.found_first %},{% endif -%}
+                            {%- set ns_items.found_first = true -%}
+                            {%- if item_key == 'properties' -%}
+                                properties:{
+                                {%- if item_value is mapping -%}
+                                    {{- format_parameters(item_value, value['items']['required'] | default([])) -}}
+                                {%- endif -%}
+                                }
+                            {%- elif item_key == 'required' -%}
+                                required:[
+                                {%- for req_item in item_value -%}
+                                    <escape>{{- req_item -}}<escape>
+                                    {%- if not loop.last %},{% endif -%}
+                                {%- endfor -%}
+                                ]
+                            {%- elif item_key == 'type' -%}
+                                {%- if item_value is string -%}
+                                    type:{{ format_argument(item_value | upper) }}
+                                {%- else -%}
+                                    type:{{ format_argument(item_value | map('upper') | list) }}
+                                {%- endif -%}
+                            {%- else -%}
+                                {{ item_key }}:{{ format_argument(item_value) }}
+                            {%- endif -%}
+                        {%- endif -%}
+                    {%- endfor -%}
+                    }
+                {%- endif -%}
+            {%- endif -%}
+            ,type:<escape>{{ value['type'] | upper }}<escape>}
+        {%- endif -%}
+    {%- endfor -%}
+{%- endmacro -%}
+{% macro format_function_declaration(tool_data) -%}
+declaration:{{- tool_data['function']['name'] -}}
+{description:<escape>{{- tool_data['function']['description'] -}}<escape>
+{%- set params = tool_data['function']['parameters'] -%}
+{%- if params -%}
+    ,parameters:{
+    {%- if params['properties'] -%}
+        properties:{ {{- format_parameters(params['properties'], params['required']) -}} },
+    {%- endif -%}
+    {%- if params['required'] -%}
+        required:[
+        {%- for item in params['required'] -%}
+            <escape>{{- item -}}<escape>
+            {{- ',' if not loop.last -}}
+        {%- endfor -%}
+        ],
+    {%- endif -%}
+    {%- if params['type'] -%}
+        type:<escape>{{- params['type'] | upper -}}<escape>}
+    {%- endif -%}
+{%- endif -%}
+}
+{%- endmacro -%}
+{% macro format_argument(argument, escape_keys=True) -%}
+{%- if argument is string -%}
+    {{- '<escape>' + argument + '<escape>' -}}
+{%- elif argument is boolean -%}
+    {%- if argument -%}
+        {{- 'true' -}}
+    {%- else -%}
+        {{- 'false' -}}
+    {%- endif -%}
+{%- elif argument is mapping -%}
+    {{- '{' -}}
+    {%- set ns = namespace(found_first=false) -%}
+    {%- for key, value in argument | dictsort -%}
+        {%- if ns.found_first %},{% endif -%}
+        {%- set ns.found_first = true -%}
+        {%- if escape_keys -%}
+            {{- '<escape>' + key + '<escape>' -}}
+        {%- else -%}
+            {{- key -}}
+        {%- endif -%}
+        :{{- format_argument(value, escape_keys=escape_keys) -}}
+    {%- endfor -%}
+    {{- '}' -}}
+{%- elif argument is sequence -%}
+    {{- '[' -}}
+    {%- for item in argument -%}
+        {{- format_argument(item, escape_keys=escape_keys) -}}
+        {%- if not loop.last %},{% endif -%}
+    {%- endfor -%}
+    {{- ']' -}}
+{%- else -%}
+    {{- argument -}}
+{%- endif -%}
+{%- endmacro -%}
+{{ bos_token }}
+{%- set ns = namespace(prev_message_type=None) -%}
+{#- Tool Declarations -#}
+{%- set loop_messages = messages -%}
+{%- if tools or messages[0]['role'] == 'system' or messages[0]['role'] == 'developer' -%}
+    {{- '<start_of_turn>developer\n' -}}
+    {%- if messages[0]['role'] == 'system' or messages[0]['role'] == 'developer' -%}
+        {%- if messages[0]['content'] is string -%}
+            {{- messages[0]['content'] | trim -}}
+        {%- elif messages[0]['content'] is sequence -%}
+            {%- for item in messages[0]['content'] -%}
+                {%- if item['type'] == 'text' -%}
+                    {{- item['text'] | trim -}}
+                {%- endif -%}
+            {%- endfor -%}
+        {%- endif -%}
+        {%- set loop_messages = messages[1:] -%}
+    {%- endif -%}
+    {%- if tools -%}
+        {%- for tool in tools %}
+            {{- '<start_function_declaration>' -}}
+            {{- format_function_declaration(tool) | trim }}
+            {{- '<end_function_declaration>' -}}
+        {%- endfor %}
+    {%- endif -%}
+    {{- '<end_of_turn>\n' }}
+{%- endif %}
+{#- Loop through messages. -#}
+{%- for message in loop_messages -%}
+    {%- if (message['role'] == 'assistant') -%}
+        {#- Rename "assistant" to "model". -#}
+        {%- set role = "model" -%}
+    {%- else -%}
+        {%- set role = message['role'] -%}
+    {%- endif -%}
+    {%- if role != 'tool' -%}
+        {%- if ns.prev_message_type != 'tool_response' -%}
+            {{- '<start_of_turn>' + role + '\n' }}
+        {%- endif -%}
+        {%- set ns.prev_message_type = None -%}
+        {%- if 'content' in message and message['content'] is not none -%}
+            {%- if message['content'] is string -%}
+                {{ message['content'] | trim }}
+            {%- elif message['content'] is sequence -%}
+                {%- for item in message['content'] -%}
+                    {%- if item['type'] == 'image' -%}
+                        {{ '<start_of_image>' }}
+                    {%- elif item['type'] == 'text' -%}
+                        {{ item['text'] | trim }}
+                    {%- endif -%}
+                {%- endfor -%}
+            {%- else -%}
+                {{ raise_exception("Invalid content type in user/assistant message") }}
+            {%- endif -%}
+            {%- set ns.prev_message_type = 'content' -%}
+        {%- endif -%}
+        {%- if 'tool_calls' in message and message['tool_calls'] and message['tool_calls'] is iterable -%}
+            {#- Tool Calls -#}
+            {%- for tool_call in message['tool_calls'] -%}
+                {% set function = tool_call['function'] %}
+                {{-  '<start_function_call>call:' + function['name'] + '{' -}}
+                {%- if 'arguments' in function -%}
+                    {%- if function['arguments'] is mapping -%}
+                        {%- set ns = namespace(found_first=false) -%}
+                        {%- for key, value in function['arguments'] | dictsort -%}
+                            {%- if ns.found_first %},{% endif -%}
+                            {%- set ns.found_first = true -%}
+                            {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                        {%- endfor -%}
+                    {%- elif function['arguments'] is string -%}
+                        {# This handles string-JSON, just in case #}
+                    {{ function['arguments'] }}
+                    {%- endif %}
+                {%- endif -%}
+                {{- '}<end_function_call>' -}}
+            {%- endfor -%}
+            {%- if loop.last -%}
+                {{ '<start_function_response>' }}
+            {%- endif -%}
+            {%- set ns.prev_message_type = 'tool_call' -%}
+        {%- endif -%}
+    {%- else -%}
+        {#- Tool Responses -#}
+        {%- if 'content' in message and message['content'] -%}
+            {%- if message['content'] is mapping -%}
+                {%- if 'name' in message['content'] and 'response' in message['content'] -%}
+                    {{ '<start_function_response>response:' + message['content']['name'] | trim + '{' }}
+                    {%- set response_ns = namespace(found_first=false) -%}
+                    {%- for key, value in message['content']['response'] | dictsort -%}
+                        {%- if response_ns.found_first %},{% endif -%}
+                        {%- set response_ns.found_first = true -%}
+                        {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                    {%- endfor -%}
+                    {{- '}<end_function_response>' -}}
+                {%- elif 'name' in message -%}
+                    {{ '<start_function_response>response:' + message['name'] | trim + '{' }}
+                    {%- set response_ns = namespace(found_first=false) -%}
+                    {%- for key, value in message['content'] | dictsort -%}
+                        {%- if response_ns.found_first %},{% endif -%}
+                        {%- set response_ns.found_first = true -%}
+                        {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                    {%- endfor -%}
+                    {{- '}<end_function_response>' -}}
+                {%- else -%}
+                    {{ raise_exception("Invalid tool response mapping: must contain 'name' and 'response' keys, or 'name' must be in the message.") }}
+                {%- endif -%}
+            {%- elif message['content'] is string -%}
+                {%- if 'name' in message -%}
+                     {{ '<start_function_response>response:' + message['name'] | trim + '{value:' + format_argument(message['content'], escape_keys=False) + '}<end_function_response>' }}
+                {%- else -%}
+                     {{ raise_exception("Invalid tool response: 'name' must be provided.") }}
+                {%- endif -%}
+            {%- elif message['content'] is sequence -%}
+                {%- for item in message['content'] -%}
+                    {%- if item is mapping -%}
+                        {%- if 'name' in item and 'response' in item -%}
+                            {{ '<start_function_response>response:' + item['name'] | trim + '{' }}
+                            {%- set response_ns = namespace(found_first=false) -%}
+                            {%- for key, value in item['response'] | dictsort -%}
+                                {%- if response_ns.found_first %},{% endif -%}
+                                {%- set response_ns.found_first = true -%}
+                                {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                            {%- endfor -%}
+                            {{- '}<end_function_response>' -}}
+                        {%- elif 'name' in message -%}
+                            {{ '<start_function_response>response:' + message['name'] | trim + '{' }}
+                            {%- set response_ns = namespace(found_first=false) -%}
+                            {%- for key, value in item | dictsort -%}
+                                {%- if response_ns.found_first %},{% endif -%}
+                                {%- set response_ns.found_first = true -%}
+                                {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
+                            {%- endfor -%}
+                            {{- '}<end_function_response>' -}}
+                        {%- else -%}
+                            {{ raise_exception("Invalid tool response mapping: must contain 'name' and 'response' keys, or 'name' must be in the message.") }}
+                        {%- endif -%}
+                    {%- else -%}
+                        {{ raise_exception("Invalid tool response message: multiple responses must all be mappings") }}
+                    {%- endif -%}
+                {%- endfor -%}
+            {%- else -%}
+                {{ raise_exception("Invalid content type in tool message: must be mapping, sequence of mappings, or string.") }}
+            {%- endif -%}
+        {%- endif -%}
+        {%- set ns.prev_message_type = 'tool_response' -%}
+    {%- endif -%}
+    {%- if ns.prev_message_type not in ['tool_call', 'tool_response'] -%}
+        {{ '<end_of_turn>\n' }}
+    {%- endif -%}
+{%- endfor -%}
+{%- if add_generation_prompt -%}
+    {%- if ns.prev_message_type != 'tool_response' -%}
+        {{- '<start_of_turn>model\n' -}}
+    {%- endif -%}
+{%- endif -%}

checkpoint-77/config.json ADDED Viewed

	@@ -0,0 +1,62 @@

+{
+  "_sliding_window_pattern": 6,
+  "architectures": [
+    "Gemma3ForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "attn_logit_softcapping": null,
+  "bos_token_id": 2,
+  "dtype": "bfloat16",
+  "eos_token_id": 1,
+  "final_logit_softcapping": null,
+  "head_dim": 256,
+  "hidden_activation": "gelu_pytorch_tanh",
+  "hidden_size": 640,
+  "initializer_range": 0.02,
+  "intermediate_size": 2048,
+  "layer_types": [
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention"
+  ],
+  "max_position_embeddings": 32768,
+  "model_type": "gemma3_text",
+  "num_attention_heads": 4,
+  "num_hidden_layers": 18,
+  "num_key_value_heads": 1,
+  "pad_token_id": 0,
+  "query_pre_attn_scalar": 256,
+  "rms_norm_eps": 1e-06,
+  "rope_parameters": {
+    "full_attention": {
+      "rope_theta": 1000000.0,
+      "rope_type": "default"
+    },
+    "sliding_attention": {
+      "rope_theta": 10000.0,
+      "rope_type": "default"
+    }
+  },
+  "sliding_window": 512,
+  "tie_word_embeddings": true,
+  "transformers_version": "5.5.1",
+  "use_bidirectional_attention": false,
+  "use_cache": false,
+  "vocab_size": 262144
+}

checkpoint-77/generation_config.json ADDED Viewed

	@@ -0,0 +1,15 @@

+{
+  "bos_token_id": 2,
+  "cache_implementation": "hybrid",
+  "do_sample": true,
+  "eos_token_id": [
+    1,
+    1,
+    50,
+    106
+  ],
+  "pad_token_id": 0,
+  "top_k": 64,
+  "top_p": 0.95,
+  "transformers_version": "5.5.1"
+}

checkpoint-77/model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:394256bfee05b42e5f2df4d6d70a8cc4803914120d528782a9e2906a5796970f
+size 536223056

checkpoint-77/optimizer.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9eb7d1d4bed0c5d4d8629496b7d67361f2a17dec7dd93d23d32503c6aa170495
+size 1072594443

checkpoint-77/rng_state.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:098b29492211804ab324a36f37466821d948280bb74fce4ba895c03f13ecd878
+size 14645

checkpoint-77/scheduler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:41252c4518652c4d654b361ca3ba34bc3d27477f1e95b590d2d97028117662b4
+size 1465

checkpoint-77/tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:80d7f800b949accd7eb940bac75e642f9468e4df157403032a55bf54ed23b650
+size 33384898

checkpoint-77/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,26 @@

+{
+  "backend": "tokenizers",
+  "boi_token": "<start_of_image>",
+  "bos_token": "<bos>",
+  "clean_up_tokenization_spaces": false,
+  "eoi_token": "<end_of_image>",
+  "eos_token": "<eos>",
+  "image_token": "<image_soft_token>",
+  "is_local": false,
+  "mask_token": "<mask>",
+  "model_max_length": 1000000000000000019884624838656,
+  "model_specific_special_tokens": {
+    "boi_token": "<start_of_image>",
+    "eoi_token": "<end_of_image>",
+    "image_token": "<image_soft_token>",
+    "sfr_token": "<start_function_response>"
+  },
+  "pad_token": "<pad>",
+  "padding_side": "left",
+  "sfr_token": "<start_function_response>",
+  "sp_model_kwargs": null,
+  "spaces_between_special_tokens": false,
+  "tokenizer_class": "GemmaTokenizer",
+  "unk_token": "<unk>",
+  "use_default_system_prompt": false
+}

checkpoint-77/trainer_state.json ADDED Viewed

	@@ -0,0 +1,55 @@

+{
+  "best_global_step": null,
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 1.0,
+  "eval_steps": 50,
+  "global_step": 77,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "entropy": 0.8290567851066589,
+      "epoch": 0.6568144499178982,
+      "grad_norm": 0.75390625,
+      "learning_rate": 9.388394947836278e-06,
+      "loss": 0.11062156677246093,
+      "mean_token_accuracy": 0.9778690934181213,
+      "num_tokens": 389732.0,
+      "step": 50
+    },
+    {
+      "epoch": 0.6568144499178982,
+      "eval_entropy": 0.7784549756483599,
+      "eval_loss": 0.03614399954676628,
+      "eval_mean_token_accuracy": 0.9917981918756064,
+      "eval_num_tokens": 389732.0,
+      "eval_runtime": 12.9713,
+      "eval_samples_per_second": 23.513,
+      "eval_steps_per_second": 5.936,
+      "step": 50
+    }
+  ],
+  "logging_steps": 50,
+  "max_steps": 308,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 4,
+  "save_steps": 500,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": false
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 503646432269568.0,
+  "train_batch_size": 4,
+  "trial_name": null,
+  "trial_params": null
+}

checkpoint-77/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e5363ce9d99dd9dbf1bc2c5ff6c4b0ce553147005c7381cbde8798f93d5971fb
+size 5649

config.json ADDED Viewed

	@@ -0,0 +1,62 @@

+{
+  "_sliding_window_pattern": 6,
+  "architectures": [
+    "Gemma3ForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "attn_logit_softcapping": null,
+  "bos_token_id": 2,
+  "dtype": "bfloat16",
+  "eos_token_id": 1,
+  "final_logit_softcapping": null,
+  "head_dim": 256,
+  "hidden_activation": "gelu_pytorch_tanh",
+  "hidden_size": 640,
+  "initializer_range": 0.02,
+  "intermediate_size": 2048,
+  "layer_types": [
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention"
+  ],
+  "max_position_embeddings": 32768,
+  "model_type": "gemma3_text",
+  "num_attention_heads": 4,
+  "num_hidden_layers": 18,
+  "num_key_value_heads": 1,
+  "pad_token_id": 0,
+  "query_pre_attn_scalar": 256,
+  "rms_norm_eps": 1e-06,
+  "rope_parameters": {
+    "full_attention": {
+      "rope_theta": 1000000.0,
+      "rope_type": "default"
+    },
+    "sliding_attention": {
+      "rope_theta": 10000.0,
+      "rope_type": "default"
+    }
+  },
+  "sliding_window": 512,
+  "tie_word_embeddings": true,
+  "transformers_version": "5.5.1",
+  "use_bidirectional_attention": false,
+  "use_cache": false,
+  "vocab_size": 262144
+}

generation_config.json ADDED Viewed

	@@ -0,0 +1,15 @@

+{
+  "bos_token_id": 2,
+  "cache_implementation": "hybrid",
+  "do_sample": true,
+  "eos_token_id": [
+    1,
+    1,
+    50,
+    106
+  ],
+  "pad_token_id": 0,
+  "top_k": 64,
+  "top_p": 0.95,
+  "transformers_version": "5.5.1"
+}

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:5fdad25a78c297aab9ec2b809949a6f0b1968f7a6a09d486343d48fe3f5c51da
+size 536223056