Synchronizing local compiler cache.
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +122 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/441269935591cad8d370e512c0b93cdd2fce6247c40e5a4866d872ee5338b0de/4ef38fe13fd2c7209a1f.json +81 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/gemma3_text/unsloth/gemma-3-270m-it/4ef38fe13fd2c7209a1f.json +81 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/441269935591cad8d370e512c0b93cdd2fce6247c40e5a4866d872ee5338b0de/70022928da5b1a3a3562.json +81 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/3dc71b9dfd0bde51256a.json +63 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/gemma3_text/unsloth/gemma-3-270m-it/70022928da5b1a3a3562.json +81 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/llama/unsloth/Llama-3.2-1B-Instruct/3dc71b9dfd0bde51256a.json +63 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/03b0c107d1cede36875199a5d51decfe04c473de2af9999f8577a028d74d0ab4/43349245d62eb2d76d64.json +189 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/03b0c107d1cede36875199a5d51decfe04c473de2af9999f8577a028d74d0ab4/566b663bde0d89e24b29.json +188 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/03b0c107d1cede36875199a5d51decfe04c473de2af9999f8577a028d74d0ab4/756910ddabb85196bbca.json +188 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/03b0c107d1cede36875199a5d51decfe04c473de2af9999f8577a028d74d0ab4/969ba466803204052129.json +189 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/0cc25526e2cfc37a8875a3752f33c4d7505d8a07b869d0f3f41915cf6e763b74/efa7f046361aa22fb2b9.json +58 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/0cc25526e2cfc37a8875a3752f33c4d7505d8a07b869d0f3f41915cf6e763b74/fa7d24fcf68294ebad29.json +58 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/2da7a00f0478d50ae1e7f75f085c5b2773b5f355f427c61cf34cb6febd629d96/e5ddd3afb102c02baf93.json +59 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/441269935591cad8d370e512c0b93cdd2fce6247c40e5a4866d872ee5338b0de/51c77185b9832eaebdfc.json +81 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/441269935591cad8d370e512c0b93cdd2fce6247c40e5a4866d872ee5338b0de/51eb5f08405da966cefd.json +81 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/441269935591cad8d370e512c0b93cdd2fce6247c40e5a4866d872ee5338b0de/6109da25218b5116e9b5.json +81 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/4ab8140bc7eb4a553d95855c5c2be2cf8c0fbab21b823d76183b6f51e98b6fc5/03a64a22d1b885eece61.json +58 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/4cb7aff9e2a15c151396f2b684013e39d6739f0dec83e5c9dabbfe9d5fcf77b7/1c8b4a21eb41ff235945.json +82 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/4cb7aff9e2a15c151396f2b684013e39d6739f0dec83e5c9dabbfe9d5fcf77b7/b12b8be52c487dcc560f.json +82 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/6454afdf3e9d66c7226c13a575b718845c25e53b0699600ba2bb4f883e9d841b/97e03fcfc7f46ac8836e.json +62 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/73707b485eab9008c7aba7f5dad0ce2384ac685318d5f888c12fa0d81ed90b19/078d41a850db4b8221d6.json +134 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/73707b485eab9008c7aba7f5dad0ce2384ac685318d5f888c12fa0d81ed90b19/2dc771f3c35b9af34ce4.json +134 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/73707b485eab9008c7aba7f5dad0ce2384ac685318d5f888c12fa0d81ed90b19/d04c2a3f18746af5f901.json +134 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/7518518c7e077820070186deda960d8cc49db068cdf0ac70664098fa2b6b698c/29c61ad2f54baaec4c1d.json +64 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/7f05bde17c7b0ffeb657897697f23d182f406b76ced7f1b2cd5741dc93fe2e2e/dadf01a9f544218eabdd.json +125 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/8c90ac2593ed0b7f1ecb60e82cb184fb11f2ea640befa1cc7b10766a5c02525d/3c497b71919d297a9da9.json +164 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/8c90ac2593ed0b7f1ecb60e82cb184fb11f2ea640befa1cc7b10766a5c02525d/d00ca5f7300e7c2696dd.json +164 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/920f44ce6d3e004d1ce547ae06644f7be262180644b04573153aa15d98742edc/b5678f2b1f926f36a4fd.json +65 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/929b02754a13cbfdf657d863c3fc6f3bce672879bc6ae48ab45be21e881e9ec2/21441e8ed07d8b61298d.json +87 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/929b02754a13cbfdf657d863c3fc6f3bce672879bc6ae48ab45be21e881e9ec2/f65f144a780153ddd757.json +87 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/4d99c6a74830655285f6.json +63 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/cb2c5b9dc81e576d83aa.json +63 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/d020f018e0819410feb2.json +63 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/da4adf3105368a5df618.json +63 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/d139acf64685f15794bb983ff6eb881bdd31304bae88b0ce1ed20a54c21f2265/0e7a6f2933f99785cba6.json +58 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/gemma3_text/unsloth/gemma-3-270m-it/51c77185b9832eaebdfc.json +81 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/granite/hf-internal-testing/tiny-random-GraniteForCausalLM/0e7a6f2933f99785cba6.json +58 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/granite/ibm-granite/granite-3.1-2b-instruct/efa7f046361aa22fb2b9.json +58 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/idefics3/HuggingFaceTB/SmolVLM-256M-Instruct/566b663bde0d89e24b29.json +188 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/llama/llamafactory/tiny-random-Llama-3/97e03fcfc7f46ac8836e.json +62 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/llama/unsloth/Llama-3.2-1B-Instruct/4d99c6a74830655285f6.json +63 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/llama4/tiny-random/llama-4/dadf01a9f544218eabdd.json +125 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/mixtral/dacorvo/Mixtral-tiny/03a64a22d1b885eece61.json +58 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/phi3/microsoft/Phi-3.5-mini-instruct/d00ca5f7300e7c2696dd.json +164 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/phi3/yujiepan/phi-4-tiny-random/e5ddd3afb102c02baf93.json +59 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/qwen2/Qwen/Qwen2.5-0.5B/1c8b4a21eb41ff235945.json +82 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/qwen2/yujiepan/qwen2.5-128k-tiny-random/29c61ad2f54baaec4c1d.json +64 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/qwen3/Qwen/Qwen3-0.6B/f65f144a780153ddd757.json +87 -0
- neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/qwen3_moe/optimum-internal-testing/tiny-random-qwen3_moe/b5678f2b1f926f36a4fd.json +65 -0
.gitattributes
CHANGED
|
@@ -17414,3 +17414,125 @@ neuronxcc-2.23.6484.0+3b612583/MODULE_17549650398225281442+f7f529f3/model.neff f
|
|
| 17414 |
neuronxcc-2.23.6484.0+3b612583/MODULE_2912675361925083263+f7f529f3/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17415 |
neuronxcc-2.23.6484.0+3b612583/MODULE_6870343675213383355+f7f529f3/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17416 |
neuronxcc-2.23.6484.0+3b612583/MODULE_9355619343028815470+f7f529f3/model.neff filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 17414 |
neuronxcc-2.23.6484.0+3b612583/MODULE_2912675361925083263+f7f529f3/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17415 |
neuronxcc-2.23.6484.0+3b612583/MODULE_6870343675213383355+f7f529f3/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17416 |
neuronxcc-2.23.6484.0+3b612583/MODULE_9355619343028815470+f7f529f3/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17417 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_083c75747563fca496d7+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17418 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_0cb2d9cedfdb8f619286+2e929b57/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17419 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_105c0ff965441eae0a1c+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17420 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_12e7cddd1a2b1bbf480f+6170d8e1/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17421 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_134003a684b11c49131d+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17422 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_15503648c89ebbfa6dc4+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17423 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_1662e89e76097dde374a+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17424 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_17bab0cf99484f23e361+b02446f6/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17425 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_17bab0cf99484f23e361+b02446f6/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17426 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_18b928d4e4846c6ce9c7+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17427 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_1b43d7c01692b5e7842e+9e8e849c/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17428 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_1b9cad3a2eb3c406661d+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17429 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_1b9cad3a2eb3c406661d+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17430 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_1f6ea746117003b733f2+b02446f6/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17431 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_1f6ea746117003b733f2+b02446f6/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17432 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_213ad9728f93ce707ff0+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17433 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_2a8f4b110978c8fcb167+c4f887dc/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17434 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_2a8f4b110978c8fcb167+c4f887dc/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17435 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_303a765b958c392446f2+c4f887dc/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17436 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_303a765b958c392446f2+c4f887dc/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17437 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_3164b243bf918aa0b477+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17438 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_31af602603367560091b+4070fd2c/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17439 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_33dcb1875eda19b3ebb5+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17440 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_3fa54d08f9bf0d057baa+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17441 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_3fa54d08f9bf0d057baa+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17442 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_43dbb7c2a39f5c4fb1c8+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17443 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_43dbb7c2a39f5c4fb1c8+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17444 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_4978e0dffee300919f4f+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17445 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_4dd21482986783a76fee+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17446 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_4f8ed259d620f5f869d0+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17447 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_4f8ed259d620f5f869d0+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17448 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_50933980a498d1588ed1+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17449 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_50933980a498d1588ed1+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17450 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_531086d2d1e6a2d23ebd+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17451 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_560b92ed487f36488c4c+c4f887dc/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17452 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_560b92ed487f36488c4c+c4f887dc/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17453 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_5b12e9643c49ec4b4ebd+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17454 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_5b12e9643c49ec4b4ebd+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17455 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_5f5b72ab3e0b8d826f02+c4f887dc/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17456 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_5f5b72ab3e0b8d826f02+c4f887dc/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17457 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_6040d38a7245c853f482+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17458 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_6040d38a7245c853f482+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17459 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_61998e972adde9b6f5d2+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17460 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_633cc41552f3f933642a+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17461 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_633cc41552f3f933642a+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17462 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_6abd2a6ee6cb217440b5+c4f887dc/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17463 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_6abd2a6ee6cb217440b5+c4f887dc/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17464 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_7701466c22adfdd959a6+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17465 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_7701466c22adfdd959a6+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17466 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_78aec5e2f32f6fc831aa+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17467 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_78aec5e2f32f6fc831aa+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17468 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_7cc268c882c393abfe7c+677eeb9d/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17469 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_866615121c80b108f18d+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17470 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_866615121c80b108f18d+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17471 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_8b7c51b3642e17a79f64+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17472 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_8e252cfdf6f4e90ebf4a+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17473 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_8e252cfdf6f4e90ebf4a+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17474 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_8ff1216675c7144590c1+c4f887dc/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17475 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_8ff1216675c7144590c1+c4f887dc/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17476 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_8ff80109ec96db204c79+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17477 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_8ff80109ec96db204c79+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17478 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_933998fc47a290de4562+c4f887dc/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17479 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_933998fc47a290de4562+c4f887dc/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17480 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_952fafb4c315904dcb0e+f2c40fef/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17481 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_95f18043665401be9ab1+c4f887dc/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17482 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_95f18043665401be9ab1+c4f887dc/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17483 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_9748d4fa623593f9a070+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17484 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_9748d4fa623593f9a070+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17485 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_976a4227c74e1e5d858d+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17486 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_976a4227c74e1e5d858d+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17487 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_9b7eb59becdba58feaf4+c4f887dc/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17488 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_9b7eb59becdba58feaf4+c4f887dc/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17489 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_a02e663f9d5e1913e9e3+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17490 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_a36debd95d53c8bebd53+f7cce17f/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17491 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_a36fd4fef78c745e1416+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17492 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_a36fd4fef78c745e1416+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17493 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_a73e950210b13ba9bed1+c4f887dc/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17494 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_a73e950210b13ba9bed1+c4f887dc/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17495 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_ab7617cb5e9186411e52+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17496 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_ab7617cb5e9186411e52+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17497 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_ac3c7cc1cd56727823f0+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17498 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_ac3c7cc1cd56727823f0+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17499 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_b1639d901305de0ccbb9+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17500 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_b406585baf2c99ea743d+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17501 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_b406585baf2c99ea743d+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17502 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_b681faf194284309cdeb+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17503 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_b96a1f42beeb09ee40ca+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17504 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_c05382aac6cf9c66958b+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17505 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_c480f8583bce4a388b93+c4f887dc/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17506 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_c480f8583bce4a388b93+c4f887dc/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17507 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_c49ef4b16e6a8513ee7c+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17508 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_c49ef4b16e6a8513ee7c+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17509 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_c5bb11161997e6aa48a1+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17510 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_c5bb11161997e6aa48a1+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17511 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_d5b145fe6e14064993e3+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17512 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_d5b145fe6e14064993e3+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17513 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_d64e5c89cbc237bea34f+b4e83f56/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17514 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_d64e5c89cbc237bea34f+b4e83f56/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17515 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_d6713fbd83cd891615f3+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17516 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_d6713fbd83cd891615f3+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17517 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_d6f75527452b7a1e9637+b02446f6/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17518 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_d6f75527452b7a1e9637+b02446f6/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17519 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_d962f9089341da526ccb+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17520 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_d962f9089341da526ccb+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17521 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_d986d392e2b714a772e0+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17522 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_da330f1b1aad44160016+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17523 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_dba668a28cef64f8b275+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17524 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_dba668a28cef64f8b275+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17525 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_dbda60963b73c3571662+186ca4ef/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17526 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_dbda60963b73c3571662+186ca4ef/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17527 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_e21228aada91cc0cee76+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17528 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_e21228aada91cc0cee76+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17529 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_ea76254177cf576ffcb2+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17530 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_eaa8316a0cd427639f95+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17531 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_eaa8316a0cd427639f95+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17532 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_ede677b769dff6a4314c+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17533 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_ee2fac66603ba635ef0e+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17534 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_ee2fac66603ba635ef0e+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17535 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_f90d09e1438492736d3c+c4f887dc/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17536 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_f90d09e1438492736d3c+c4f887dc/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
|
| 17537 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_f9154701c36b9388a53d+00ac9e50/model.neff filter=lfs diff=lfs merge=lfs -text
|
| 17538 |
+
neuronxcc-2.21.33363.0+82129205/MODULE_ff54f59684fbb72ef7e9+6170d8e1/model.neff filter=lfs diff=lfs merge=lfs -text
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/441269935591cad8d370e512c0b93cdd2fce6247c40e5a4866d872ee5338b0de/4ef38fe13fd2c7209a1f.json
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "unsloth/gemma-3-270m-it",
|
| 4 |
+
"_sliding_window_pattern": 6,
|
| 5 |
+
"_task": "text-generation",
|
| 6 |
+
"architectures": [
|
| 7 |
+
"Gemma3ForCausalLM"
|
| 8 |
+
],
|
| 9 |
+
"attention_bias": false,
|
| 10 |
+
"attention_dropout": 0.0,
|
| 11 |
+
"attn_logit_softcapping": null,
|
| 12 |
+
"dtype": "bfloat16",
|
| 13 |
+
"final_logit_softcapping": null,
|
| 14 |
+
"head_dim": 256,
|
| 15 |
+
"hidden_activation": "gelu_pytorch_tanh",
|
| 16 |
+
"hidden_size": 640,
|
| 17 |
+
"initializer_range": 0.02,
|
| 18 |
+
"intermediate_size": 2048,
|
| 19 |
+
"layer_types": [
|
| 20 |
+
"sliding_attention",
|
| 21 |
+
"sliding_attention",
|
| 22 |
+
"sliding_attention",
|
| 23 |
+
"sliding_attention",
|
| 24 |
+
"sliding_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"sliding_attention",
|
| 27 |
+
"sliding_attention",
|
| 28 |
+
"sliding_attention",
|
| 29 |
+
"sliding_attention",
|
| 30 |
+
"sliding_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"sliding_attention",
|
| 33 |
+
"sliding_attention",
|
| 34 |
+
"sliding_attention",
|
| 35 |
+
"sliding_attention",
|
| 36 |
+
"sliding_attention",
|
| 37 |
+
"full_attention"
|
| 38 |
+
],
|
| 39 |
+
"max_position_embeddings": 32768,
|
| 40 |
+
"model_type": "gemma3_text",
|
| 41 |
+
"neuron": {
|
| 42 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 43 |
+
"batch_size": 1,
|
| 44 |
+
"capacity_factor": null,
|
| 45 |
+
"checkpoint_id": "unsloth/gemma-3-270m-it",
|
| 46 |
+
"checkpoint_revision": "23cf460f6bb16954176b3ddcc8d4f250501458a9",
|
| 47 |
+
"continuous_batching": false,
|
| 48 |
+
"ep_degree": 1,
|
| 49 |
+
"fused_qkv": true,
|
| 50 |
+
"glu_mlp": true,
|
| 51 |
+
"local_ranks_size": 2,
|
| 52 |
+
"max_batch_size": 1,
|
| 53 |
+
"max_context_length": 8192,
|
| 54 |
+
"max_topk": 256,
|
| 55 |
+
"n_active_tokens": 8192,
|
| 56 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 57 |
+
"on_device_sampling": true,
|
| 58 |
+
"optimum_neuron_version": "0.4.6.dev1",
|
| 59 |
+
"output_logits": false,
|
| 60 |
+
"pp_degree": 1,
|
| 61 |
+
"sequence_length": 8192,
|
| 62 |
+
"speculation_length": 0,
|
| 63 |
+
"start_rank_id": 0,
|
| 64 |
+
"target": "trn1",
|
| 65 |
+
"torch_dtype": "bfloat16",
|
| 66 |
+
"tp_degree": 2
|
| 67 |
+
},
|
| 68 |
+
"num_attention_heads": 4,
|
| 69 |
+
"num_hidden_layers": 18,
|
| 70 |
+
"num_key_value_heads": 1,
|
| 71 |
+
"query_pre_attn_scalar": 256,
|
| 72 |
+
"rms_norm_eps": 1e-06,
|
| 73 |
+
"rope_local_base_freq": 10000.0,
|
| 74 |
+
"rope_scaling": null,
|
| 75 |
+
"rope_theta": 1000000.0,
|
| 76 |
+
"sliding_window": 512,
|
| 77 |
+
"unsloth_fixed": true,
|
| 78 |
+
"use_bidirectional_attention": false,
|
| 79 |
+
"use_cache": true,
|
| 80 |
+
"vocab_size": 262144
|
| 81 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/gemma3_text/unsloth/gemma-3-270m-it/4ef38fe13fd2c7209a1f.json
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "unsloth/gemma-3-270m-it",
|
| 4 |
+
"_sliding_window_pattern": 6,
|
| 5 |
+
"_task": "text-generation",
|
| 6 |
+
"architectures": [
|
| 7 |
+
"Gemma3ForCausalLM"
|
| 8 |
+
],
|
| 9 |
+
"attention_bias": false,
|
| 10 |
+
"attention_dropout": 0.0,
|
| 11 |
+
"attn_logit_softcapping": null,
|
| 12 |
+
"dtype": "bfloat16",
|
| 13 |
+
"final_logit_softcapping": null,
|
| 14 |
+
"head_dim": 256,
|
| 15 |
+
"hidden_activation": "gelu_pytorch_tanh",
|
| 16 |
+
"hidden_size": 640,
|
| 17 |
+
"initializer_range": 0.02,
|
| 18 |
+
"intermediate_size": 2048,
|
| 19 |
+
"layer_types": [
|
| 20 |
+
"sliding_attention",
|
| 21 |
+
"sliding_attention",
|
| 22 |
+
"sliding_attention",
|
| 23 |
+
"sliding_attention",
|
| 24 |
+
"sliding_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"sliding_attention",
|
| 27 |
+
"sliding_attention",
|
| 28 |
+
"sliding_attention",
|
| 29 |
+
"sliding_attention",
|
| 30 |
+
"sliding_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"sliding_attention",
|
| 33 |
+
"sliding_attention",
|
| 34 |
+
"sliding_attention",
|
| 35 |
+
"sliding_attention",
|
| 36 |
+
"sliding_attention",
|
| 37 |
+
"full_attention"
|
| 38 |
+
],
|
| 39 |
+
"max_position_embeddings": 32768,
|
| 40 |
+
"model_type": "gemma3_text",
|
| 41 |
+
"neuron": {
|
| 42 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 43 |
+
"batch_size": 1,
|
| 44 |
+
"capacity_factor": null,
|
| 45 |
+
"checkpoint_id": "unsloth/gemma-3-270m-it",
|
| 46 |
+
"checkpoint_revision": "23cf460f6bb16954176b3ddcc8d4f250501458a9",
|
| 47 |
+
"continuous_batching": false,
|
| 48 |
+
"ep_degree": 1,
|
| 49 |
+
"fused_qkv": true,
|
| 50 |
+
"glu_mlp": true,
|
| 51 |
+
"local_ranks_size": 2,
|
| 52 |
+
"max_batch_size": 1,
|
| 53 |
+
"max_context_length": 8192,
|
| 54 |
+
"max_topk": 256,
|
| 55 |
+
"n_active_tokens": 8192,
|
| 56 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 57 |
+
"on_device_sampling": true,
|
| 58 |
+
"optimum_neuron_version": "0.4.6.dev1",
|
| 59 |
+
"output_logits": false,
|
| 60 |
+
"pp_degree": 1,
|
| 61 |
+
"sequence_length": 8192,
|
| 62 |
+
"speculation_length": 0,
|
| 63 |
+
"start_rank_id": 0,
|
| 64 |
+
"target": "trn1",
|
| 65 |
+
"torch_dtype": "bfloat16",
|
| 66 |
+
"tp_degree": 2
|
| 67 |
+
},
|
| 68 |
+
"num_attention_heads": 4,
|
| 69 |
+
"num_hidden_layers": 18,
|
| 70 |
+
"num_key_value_heads": 1,
|
| 71 |
+
"query_pre_attn_scalar": 256,
|
| 72 |
+
"rms_norm_eps": 1e-06,
|
| 73 |
+
"rope_local_base_freq": 10000.0,
|
| 74 |
+
"rope_scaling": null,
|
| 75 |
+
"rope_theta": 1000000.0,
|
| 76 |
+
"sliding_window": 512,
|
| 77 |
+
"unsloth_fixed": true,
|
| 78 |
+
"use_bidirectional_attention": false,
|
| 79 |
+
"use_cache": true,
|
| 80 |
+
"vocab_size": 262144
|
| 81 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/441269935591cad8d370e512c0b93cdd2fce6247c40e5a4866d872ee5338b0de/70022928da5b1a3a3562.json
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "unsloth/gemma-3-270m-it",
|
| 4 |
+
"_sliding_window_pattern": 6,
|
| 5 |
+
"_task": "text-generation",
|
| 6 |
+
"architectures": [
|
| 7 |
+
"Gemma3ForCausalLM"
|
| 8 |
+
],
|
| 9 |
+
"attention_bias": false,
|
| 10 |
+
"attention_dropout": 0.0,
|
| 11 |
+
"attn_logit_softcapping": null,
|
| 12 |
+
"dtype": "bfloat16",
|
| 13 |
+
"final_logit_softcapping": null,
|
| 14 |
+
"head_dim": 256,
|
| 15 |
+
"hidden_activation": "gelu_pytorch_tanh",
|
| 16 |
+
"hidden_size": 640,
|
| 17 |
+
"initializer_range": 0.02,
|
| 18 |
+
"intermediate_size": 2048,
|
| 19 |
+
"layer_types": [
|
| 20 |
+
"sliding_attention",
|
| 21 |
+
"sliding_attention",
|
| 22 |
+
"sliding_attention",
|
| 23 |
+
"sliding_attention",
|
| 24 |
+
"sliding_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"sliding_attention",
|
| 27 |
+
"sliding_attention",
|
| 28 |
+
"sliding_attention",
|
| 29 |
+
"sliding_attention",
|
| 30 |
+
"sliding_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"sliding_attention",
|
| 33 |
+
"sliding_attention",
|
| 34 |
+
"sliding_attention",
|
| 35 |
+
"sliding_attention",
|
| 36 |
+
"sliding_attention",
|
| 37 |
+
"full_attention"
|
| 38 |
+
],
|
| 39 |
+
"max_position_embeddings": 32768,
|
| 40 |
+
"model_type": "gemma3_text",
|
| 41 |
+
"neuron": {
|
| 42 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 43 |
+
"batch_size": 1,
|
| 44 |
+
"capacity_factor": null,
|
| 45 |
+
"checkpoint_id": "unsloth/gemma-3-270m-it",
|
| 46 |
+
"checkpoint_revision": "23cf460f6bb16954176b3ddcc8d4f250501458a9",
|
| 47 |
+
"continuous_batching": false,
|
| 48 |
+
"ep_degree": 1,
|
| 49 |
+
"fused_qkv": true,
|
| 50 |
+
"glu_mlp": true,
|
| 51 |
+
"local_ranks_size": 2,
|
| 52 |
+
"max_batch_size": 1,
|
| 53 |
+
"max_context_length": 8192,
|
| 54 |
+
"max_topk": 256,
|
| 55 |
+
"n_active_tokens": 8192,
|
| 56 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 57 |
+
"on_device_sampling": true,
|
| 58 |
+
"optimum_neuron_version": "0.4.6.dev2",
|
| 59 |
+
"output_logits": false,
|
| 60 |
+
"pp_degree": 1,
|
| 61 |
+
"sequence_length": 8192,
|
| 62 |
+
"speculation_length": 0,
|
| 63 |
+
"start_rank_id": 0,
|
| 64 |
+
"target": "trn1",
|
| 65 |
+
"torch_dtype": "bfloat16",
|
| 66 |
+
"tp_degree": 2
|
| 67 |
+
},
|
| 68 |
+
"num_attention_heads": 4,
|
| 69 |
+
"num_hidden_layers": 18,
|
| 70 |
+
"num_key_value_heads": 1,
|
| 71 |
+
"query_pre_attn_scalar": 256,
|
| 72 |
+
"rms_norm_eps": 1e-06,
|
| 73 |
+
"rope_local_base_freq": 10000.0,
|
| 74 |
+
"rope_scaling": null,
|
| 75 |
+
"rope_theta": 1000000.0,
|
| 76 |
+
"sliding_window": 512,
|
| 77 |
+
"unsloth_fixed": true,
|
| 78 |
+
"use_bidirectional_attention": false,
|
| 79 |
+
"use_cache": true,
|
| 80 |
+
"vocab_size": 262144
|
| 81 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/3dc71b9dfd0bde51256a.json
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "unsloth/Llama-3.2-1B-Instruct",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"LlamaForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 64,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 2048,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 8192,
|
| 16 |
+
"max_position_embeddings": 131072,
|
| 17 |
+
"mlp_bias": false,
|
| 18 |
+
"model_type": "llama",
|
| 19 |
+
"neuron": {
|
| 20 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 21 |
+
"batch_size": 1,
|
| 22 |
+
"capacity_factor": null,
|
| 23 |
+
"checkpoint_id": "unsloth/Llama-3.2-1B-Instruct",
|
| 24 |
+
"checkpoint_revision": "5a8abab4a5d6f164389b1079fb721cfab8d7126c",
|
| 25 |
+
"continuous_batching": false,
|
| 26 |
+
"ep_degree": 1,
|
| 27 |
+
"fused_qkv": true,
|
| 28 |
+
"glu_mlp": true,
|
| 29 |
+
"local_ranks_size": 2,
|
| 30 |
+
"max_batch_size": 1,
|
| 31 |
+
"max_context_length": 8192,
|
| 32 |
+
"max_topk": 256,
|
| 33 |
+
"n_active_tokens": 8192,
|
| 34 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 35 |
+
"on_device_sampling": true,
|
| 36 |
+
"optimum_neuron_version": "0.4.6.dev2",
|
| 37 |
+
"output_logits": false,
|
| 38 |
+
"pp_degree": 1,
|
| 39 |
+
"sequence_length": 8192,
|
| 40 |
+
"speculation_length": 0,
|
| 41 |
+
"start_rank_id": 0,
|
| 42 |
+
"target": "trn1",
|
| 43 |
+
"torch_dtype": "bfloat16",
|
| 44 |
+
"tp_degree": 2
|
| 45 |
+
},
|
| 46 |
+
"num_attention_heads": 32,
|
| 47 |
+
"num_hidden_layers": 16,
|
| 48 |
+
"num_key_value_heads": 8,
|
| 49 |
+
"pretraining_tp": 1,
|
| 50 |
+
"rms_norm_eps": 1e-05,
|
| 51 |
+
"rope_scaling": {
|
| 52 |
+
"factor": 32.0,
|
| 53 |
+
"high_freq_factor": 4.0,
|
| 54 |
+
"low_freq_factor": 1.0,
|
| 55 |
+
"original_max_position_embeddings": 8192,
|
| 56 |
+
"rope_type": "llama3"
|
| 57 |
+
},
|
| 58 |
+
"rope_theta": 500000.0,
|
| 59 |
+
"tie_word_embeddings": true,
|
| 60 |
+
"unsloth_fixed": true,
|
| 61 |
+
"use_cache": true,
|
| 62 |
+
"vocab_size": 128256
|
| 63 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/gemma3_text/unsloth/gemma-3-270m-it/70022928da5b1a3a3562.json
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "unsloth/gemma-3-270m-it",
|
| 4 |
+
"_sliding_window_pattern": 6,
|
| 5 |
+
"_task": "text-generation",
|
| 6 |
+
"architectures": [
|
| 7 |
+
"Gemma3ForCausalLM"
|
| 8 |
+
],
|
| 9 |
+
"attention_bias": false,
|
| 10 |
+
"attention_dropout": 0.0,
|
| 11 |
+
"attn_logit_softcapping": null,
|
| 12 |
+
"dtype": "bfloat16",
|
| 13 |
+
"final_logit_softcapping": null,
|
| 14 |
+
"head_dim": 256,
|
| 15 |
+
"hidden_activation": "gelu_pytorch_tanh",
|
| 16 |
+
"hidden_size": 640,
|
| 17 |
+
"initializer_range": 0.02,
|
| 18 |
+
"intermediate_size": 2048,
|
| 19 |
+
"layer_types": [
|
| 20 |
+
"sliding_attention",
|
| 21 |
+
"sliding_attention",
|
| 22 |
+
"sliding_attention",
|
| 23 |
+
"sliding_attention",
|
| 24 |
+
"sliding_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"sliding_attention",
|
| 27 |
+
"sliding_attention",
|
| 28 |
+
"sliding_attention",
|
| 29 |
+
"sliding_attention",
|
| 30 |
+
"sliding_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"sliding_attention",
|
| 33 |
+
"sliding_attention",
|
| 34 |
+
"sliding_attention",
|
| 35 |
+
"sliding_attention",
|
| 36 |
+
"sliding_attention",
|
| 37 |
+
"full_attention"
|
| 38 |
+
],
|
| 39 |
+
"max_position_embeddings": 32768,
|
| 40 |
+
"model_type": "gemma3_text",
|
| 41 |
+
"neuron": {
|
| 42 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 43 |
+
"batch_size": 1,
|
| 44 |
+
"capacity_factor": null,
|
| 45 |
+
"checkpoint_id": "unsloth/gemma-3-270m-it",
|
| 46 |
+
"checkpoint_revision": "23cf460f6bb16954176b3ddcc8d4f250501458a9",
|
| 47 |
+
"continuous_batching": false,
|
| 48 |
+
"ep_degree": 1,
|
| 49 |
+
"fused_qkv": true,
|
| 50 |
+
"glu_mlp": true,
|
| 51 |
+
"local_ranks_size": 2,
|
| 52 |
+
"max_batch_size": 1,
|
| 53 |
+
"max_context_length": 8192,
|
| 54 |
+
"max_topk": 256,
|
| 55 |
+
"n_active_tokens": 8192,
|
| 56 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 57 |
+
"on_device_sampling": true,
|
| 58 |
+
"optimum_neuron_version": "0.4.6.dev2",
|
| 59 |
+
"output_logits": false,
|
| 60 |
+
"pp_degree": 1,
|
| 61 |
+
"sequence_length": 8192,
|
| 62 |
+
"speculation_length": 0,
|
| 63 |
+
"start_rank_id": 0,
|
| 64 |
+
"target": "trn1",
|
| 65 |
+
"torch_dtype": "bfloat16",
|
| 66 |
+
"tp_degree": 2
|
| 67 |
+
},
|
| 68 |
+
"num_attention_heads": 4,
|
| 69 |
+
"num_hidden_layers": 18,
|
| 70 |
+
"num_key_value_heads": 1,
|
| 71 |
+
"query_pre_attn_scalar": 256,
|
| 72 |
+
"rms_norm_eps": 1e-06,
|
| 73 |
+
"rope_local_base_freq": 10000.0,
|
| 74 |
+
"rope_scaling": null,
|
| 75 |
+
"rope_theta": 1000000.0,
|
| 76 |
+
"sliding_window": 512,
|
| 77 |
+
"unsloth_fixed": true,
|
| 78 |
+
"use_bidirectional_attention": false,
|
| 79 |
+
"use_cache": true,
|
| 80 |
+
"vocab_size": 262144
|
| 81 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/llama/unsloth/Llama-3.2-1B-Instruct/3dc71b9dfd0bde51256a.json
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "unsloth/Llama-3.2-1B-Instruct",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"LlamaForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 64,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 2048,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 8192,
|
| 16 |
+
"max_position_embeddings": 131072,
|
| 17 |
+
"mlp_bias": false,
|
| 18 |
+
"model_type": "llama",
|
| 19 |
+
"neuron": {
|
| 20 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 21 |
+
"batch_size": 1,
|
| 22 |
+
"capacity_factor": null,
|
| 23 |
+
"checkpoint_id": "unsloth/Llama-3.2-1B-Instruct",
|
| 24 |
+
"checkpoint_revision": "5a8abab4a5d6f164389b1079fb721cfab8d7126c",
|
| 25 |
+
"continuous_batching": false,
|
| 26 |
+
"ep_degree": 1,
|
| 27 |
+
"fused_qkv": true,
|
| 28 |
+
"glu_mlp": true,
|
| 29 |
+
"local_ranks_size": 2,
|
| 30 |
+
"max_batch_size": 1,
|
| 31 |
+
"max_context_length": 8192,
|
| 32 |
+
"max_topk": 256,
|
| 33 |
+
"n_active_tokens": 8192,
|
| 34 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 35 |
+
"on_device_sampling": true,
|
| 36 |
+
"optimum_neuron_version": "0.4.6.dev2",
|
| 37 |
+
"output_logits": false,
|
| 38 |
+
"pp_degree": 1,
|
| 39 |
+
"sequence_length": 8192,
|
| 40 |
+
"speculation_length": 0,
|
| 41 |
+
"start_rank_id": 0,
|
| 42 |
+
"target": "trn1",
|
| 43 |
+
"torch_dtype": "bfloat16",
|
| 44 |
+
"tp_degree": 2
|
| 45 |
+
},
|
| 46 |
+
"num_attention_heads": 32,
|
| 47 |
+
"num_hidden_layers": 16,
|
| 48 |
+
"num_key_value_heads": 8,
|
| 49 |
+
"pretraining_tp": 1,
|
| 50 |
+
"rms_norm_eps": 1e-05,
|
| 51 |
+
"rope_scaling": {
|
| 52 |
+
"factor": 32.0,
|
| 53 |
+
"high_freq_factor": 4.0,
|
| 54 |
+
"low_freq_factor": 1.0,
|
| 55 |
+
"original_max_position_embeddings": 8192,
|
| 56 |
+
"rope_type": "llama3"
|
| 57 |
+
},
|
| 58 |
+
"rope_theta": 500000.0,
|
| 59 |
+
"tie_word_embeddings": true,
|
| 60 |
+
"unsloth_fixed": true,
|
| 61 |
+
"use_cache": true,
|
| 62 |
+
"vocab_size": 128256
|
| 63 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/03b0c107d1cede36875199a5d51decfe04c473de2af9999f8577a028d74d0ab4/43349245d62eb2d76d64.json
ADDED
|
@@ -0,0 +1,189 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "HuggingFaceTB/SmolVLM-256M-Instruct",
|
| 4 |
+
"_task": "image-text-to-text",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Idefics3ForConditionalGeneration"
|
| 7 |
+
],
|
| 8 |
+
"dtype": "bfloat16",
|
| 9 |
+
"image_token_id": 49190,
|
| 10 |
+
"model_type": "idefics3",
|
| 11 |
+
"neuron": {
|
| 12 |
+
"_serialized_key": "NxDVLMNeuronConfig",
|
| 13 |
+
"batch_size": 1,
|
| 14 |
+
"capacity_factor": null,
|
| 15 |
+
"checkpoint_id": "HuggingFaceTB/SmolVLM-256M-Instruct",
|
| 16 |
+
"checkpoint_revision": "7e3e67edbbed1bf9888184d9df282b700a323964",
|
| 17 |
+
"continuous_batching": false,
|
| 18 |
+
"ep_degree": 1,
|
| 19 |
+
"fused_qkv": true,
|
| 20 |
+
"glu_mlp": true,
|
| 21 |
+
"image_seq_len": 64,
|
| 22 |
+
"image_size": 512,
|
| 23 |
+
"local_ranks_size": 2,
|
| 24 |
+
"max_batch_size": 1,
|
| 25 |
+
"max_context_length": 2048,
|
| 26 |
+
"max_num_images": 17,
|
| 27 |
+
"max_topk": 256,
|
| 28 |
+
"n_active_tokens": 2048,
|
| 29 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 30 |
+
"on_device_sampling": true,
|
| 31 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 32 |
+
"output_logits": false,
|
| 33 |
+
"pp_degree": 1,
|
| 34 |
+
"prefill_chunk_size": 0,
|
| 35 |
+
"sequence_length": 2048,
|
| 36 |
+
"speculation_length": 0,
|
| 37 |
+
"start_rank_id": 0,
|
| 38 |
+
"target": "trn1",
|
| 39 |
+
"torch_dtype": "bfloat16",
|
| 40 |
+
"tp_degree": 2
|
| 41 |
+
},
|
| 42 |
+
"scale_factor": 4,
|
| 43 |
+
"text_config": {
|
| 44 |
+
"_attn_implementation_autoset": false,
|
| 45 |
+
"_flash_attn_2_enabled": true,
|
| 46 |
+
"_name_or_path": "None",
|
| 47 |
+
"architectures": [
|
| 48 |
+
"VLlama3ForCausalLM"
|
| 49 |
+
],
|
| 50 |
+
"attention_bias": false,
|
| 51 |
+
"attention_dropout": 0.0,
|
| 52 |
+
"dtype": "bfloat16",
|
| 53 |
+
"head_dim": 64,
|
| 54 |
+
"hidden_act": "silu",
|
| 55 |
+
"hidden_size": 576,
|
| 56 |
+
"initializer_range": 0.041666666666666664,
|
| 57 |
+
"intermediate_size": 1536,
|
| 58 |
+
"is_llama_config": true,
|
| 59 |
+
"max_position_embeddings": 8192,
|
| 60 |
+
"mlp_bias": false,
|
| 61 |
+
"model_type": "llama",
|
| 62 |
+
"neftune_noise_alpha": 0.0,
|
| 63 |
+
"num_attention_heads": 9,
|
| 64 |
+
"num_hidden_layers": 30,
|
| 65 |
+
"num_key_value_heads": 3,
|
| 66 |
+
"pad_token_id": 2,
|
| 67 |
+
"perceiver_config": {
|
| 68 |
+
"_attn_implementation_autoset": false,
|
| 69 |
+
"_name_or_path": "",
|
| 70 |
+
"add_cross_attention": false,
|
| 71 |
+
"architectures": null,
|
| 72 |
+
"attention_dropout": 0.0,
|
| 73 |
+
"bad_words_ids": null,
|
| 74 |
+
"begin_suppress_tokens": null,
|
| 75 |
+
"bos_token_id": null,
|
| 76 |
+
"chunk_size_feed_forward": 0,
|
| 77 |
+
"cross_attention_hidden_size": null,
|
| 78 |
+
"decoder_start_token_id": null,
|
| 79 |
+
"diversity_penalty": 0.0,
|
| 80 |
+
"do_sample": false,
|
| 81 |
+
"early_stopping": false,
|
| 82 |
+
"encoder_no_repeat_ngram_size": 0,
|
| 83 |
+
"eos_token_id": null,
|
| 84 |
+
"exponential_decay_length_penalty": null,
|
| 85 |
+
"finetuning_task": null,
|
| 86 |
+
"forced_bos_token_id": null,
|
| 87 |
+
"forced_eos_token_id": null,
|
| 88 |
+
"hidden_act": "silu",
|
| 89 |
+
"id2label": {
|
| 90 |
+
"0": "LABEL_0",
|
| 91 |
+
"1": "LABEL_1"
|
| 92 |
+
},
|
| 93 |
+
"is_decoder": false,
|
| 94 |
+
"is_encoder_decoder": false,
|
| 95 |
+
"label2id": {
|
| 96 |
+
"LABEL_0": 0,
|
| 97 |
+
"LABEL_1": 1
|
| 98 |
+
},
|
| 99 |
+
"length_penalty": 1.0,
|
| 100 |
+
"max_length": 20,
|
| 101 |
+
"min_length": 0,
|
| 102 |
+
"model_type": "vllama3",
|
| 103 |
+
"no_repeat_ngram_size": 0,
|
| 104 |
+
"num_beam_groups": 1,
|
| 105 |
+
"num_beams": 1,
|
| 106 |
+
"num_key_value_heads": 1,
|
| 107 |
+
"num_return_sequences": 1,
|
| 108 |
+
"output_attentions": false,
|
| 109 |
+
"output_hidden_states": false,
|
| 110 |
+
"output_scores": false,
|
| 111 |
+
"pad_token_id": null,
|
| 112 |
+
"prefix": null,
|
| 113 |
+
"problem_type": null,
|
| 114 |
+
"pruned_heads": {},
|
| 115 |
+
"qk_layer_norms_perceiver": false,
|
| 116 |
+
"remove_invalid_values": false,
|
| 117 |
+
"repetition_penalty": 1.0,
|
| 118 |
+
"resampler_depth": 6,
|
| 119 |
+
"resampler_head_dim": 96,
|
| 120 |
+
"resampler_n_heads": 16,
|
| 121 |
+
"resampler_n_latents": 64,
|
| 122 |
+
"return_dict": true,
|
| 123 |
+
"return_dict_in_generate": false,
|
| 124 |
+
"sep_token_id": null,
|
| 125 |
+
"suppress_tokens": null,
|
| 126 |
+
"task_specific_params": null,
|
| 127 |
+
"temperature": 1.0,
|
| 128 |
+
"tf_legacy_loss": false,
|
| 129 |
+
"tie_encoder_decoder": false,
|
| 130 |
+
"tie_word_embeddings": true,
|
| 131 |
+
"tokenizer_class": null,
|
| 132 |
+
"top_k": 50,
|
| 133 |
+
"top_p": 1.0,
|
| 134 |
+
"torch_dtype": null,
|
| 135 |
+
"torchscript": false,
|
| 136 |
+
"transformers_version": "4.46.0",
|
| 137 |
+
"typical_p": 1.0,
|
| 138 |
+
"use_bfloat16": false
|
| 139 |
+
},
|
| 140 |
+
"pixel_shuffle_factor": 4,
|
| 141 |
+
"pretraining_tp": 1,
|
| 142 |
+
"qk_layer_norms": false,
|
| 143 |
+
"rms_norm_eps": 1e-05,
|
| 144 |
+
"rope_interleaved": false,
|
| 145 |
+
"rope_scaling": null,
|
| 146 |
+
"rope_theta": 100000,
|
| 147 |
+
"transformers.js_config": {
|
| 148 |
+
"kv_cache_dtype": {
|
| 149 |
+
"fp16": "float16",
|
| 150 |
+
"q4f16": "float16"
|
| 151 |
+
}
|
| 152 |
+
},
|
| 153 |
+
"use_cache": true,
|
| 154 |
+
"use_resampler": false,
|
| 155 |
+
"vocab_size": 49280
|
| 156 |
+
},
|
| 157 |
+
"tie_word_embeddings": false,
|
| 158 |
+
"transformers.js_config": {
|
| 159 |
+
"kv_cache_dtype": {
|
| 160 |
+
"fp16": "float16",
|
| 161 |
+
"q4f16": "float16"
|
| 162 |
+
}
|
| 163 |
+
},
|
| 164 |
+
"use_cache": true,
|
| 165 |
+
"vision_config": {
|
| 166 |
+
"_attn_implementation_autoset": false,
|
| 167 |
+
"attention_dropout": 0.0,
|
| 168 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 169 |
+
"hidden_size": 768,
|
| 170 |
+
"image_size": 512,
|
| 171 |
+
"initializer_range": 0.02,
|
| 172 |
+
"intermediate_size": 3072,
|
| 173 |
+
"layer_norm_eps": 1e-06,
|
| 174 |
+
"max_image_size": {
|
| 175 |
+
"longest_edge": 512
|
| 176 |
+
},
|
| 177 |
+
"model_type": "idefics3_vision",
|
| 178 |
+
"num_attention_heads": 12,
|
| 179 |
+
"num_channels": 3,
|
| 180 |
+
"num_hidden_layers": 12,
|
| 181 |
+
"patch_size": 16,
|
| 182 |
+
"size": {
|
| 183 |
+
"longest_edge": 2048
|
| 184 |
+
},
|
| 185 |
+
"tie_word_embeddings": false,
|
| 186 |
+
"use_base_siglip": true
|
| 187 |
+
},
|
| 188 |
+
"vocab_size": 49280
|
| 189 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/03b0c107d1cede36875199a5d51decfe04c473de2af9999f8577a028d74d0ab4/566b663bde0d89e24b29.json
ADDED
|
@@ -0,0 +1,188 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "HuggingFaceTB/SmolVLM-256M-Instruct",
|
| 4 |
+
"_task": "image-text-to-text",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Idefics3ForConditionalGeneration"
|
| 7 |
+
],
|
| 8 |
+
"dtype": "bfloat16",
|
| 9 |
+
"image_token_id": 49190,
|
| 10 |
+
"model_type": "idefics3",
|
| 11 |
+
"neuron": {
|
| 12 |
+
"_serialized_key": "NxDVLMNeuronConfig",
|
| 13 |
+
"batch_size": 1,
|
| 14 |
+
"capacity_factor": null,
|
| 15 |
+
"checkpoint_id": "HuggingFaceTB/SmolVLM-256M-Instruct",
|
| 16 |
+
"checkpoint_revision": "7e3e67edbbed1bf9888184d9df282b700a323964",
|
| 17 |
+
"continuous_batching": false,
|
| 18 |
+
"ep_degree": 1,
|
| 19 |
+
"fused_qkv": true,
|
| 20 |
+
"glu_mlp": true,
|
| 21 |
+
"image_seq_len": 64,
|
| 22 |
+
"image_size": 512,
|
| 23 |
+
"local_ranks_size": 2,
|
| 24 |
+
"max_batch_size": 1,
|
| 25 |
+
"max_context_length": 2048,
|
| 26 |
+
"max_num_images": 1,
|
| 27 |
+
"max_topk": 256,
|
| 28 |
+
"n_active_tokens": 2048,
|
| 29 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 30 |
+
"on_device_sampling": true,
|
| 31 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 32 |
+
"output_logits": false,
|
| 33 |
+
"pp_degree": 1,
|
| 34 |
+
"sequence_length": 2048,
|
| 35 |
+
"speculation_length": 0,
|
| 36 |
+
"start_rank_id": 0,
|
| 37 |
+
"target": "trn1",
|
| 38 |
+
"torch_dtype": "bfloat16",
|
| 39 |
+
"tp_degree": 2
|
| 40 |
+
},
|
| 41 |
+
"scale_factor": 4,
|
| 42 |
+
"text_config": {
|
| 43 |
+
"_attn_implementation_autoset": false,
|
| 44 |
+
"_flash_attn_2_enabled": true,
|
| 45 |
+
"_name_or_path": "None",
|
| 46 |
+
"architectures": [
|
| 47 |
+
"VLlama3ForCausalLM"
|
| 48 |
+
],
|
| 49 |
+
"attention_bias": false,
|
| 50 |
+
"attention_dropout": 0.0,
|
| 51 |
+
"dtype": "bfloat16",
|
| 52 |
+
"head_dim": 64,
|
| 53 |
+
"hidden_act": "silu",
|
| 54 |
+
"hidden_size": 576,
|
| 55 |
+
"initializer_range": 0.041666666666666664,
|
| 56 |
+
"intermediate_size": 1536,
|
| 57 |
+
"is_llama_config": true,
|
| 58 |
+
"max_position_embeddings": 8192,
|
| 59 |
+
"mlp_bias": false,
|
| 60 |
+
"model_type": "llama",
|
| 61 |
+
"neftune_noise_alpha": 0.0,
|
| 62 |
+
"num_attention_heads": 9,
|
| 63 |
+
"num_hidden_layers": 30,
|
| 64 |
+
"num_key_value_heads": 3,
|
| 65 |
+
"pad_token_id": 2,
|
| 66 |
+
"perceiver_config": {
|
| 67 |
+
"_attn_implementation_autoset": false,
|
| 68 |
+
"_name_or_path": "",
|
| 69 |
+
"add_cross_attention": false,
|
| 70 |
+
"architectures": null,
|
| 71 |
+
"attention_dropout": 0.0,
|
| 72 |
+
"bad_words_ids": null,
|
| 73 |
+
"begin_suppress_tokens": null,
|
| 74 |
+
"bos_token_id": null,
|
| 75 |
+
"chunk_size_feed_forward": 0,
|
| 76 |
+
"cross_attention_hidden_size": null,
|
| 77 |
+
"decoder_start_token_id": null,
|
| 78 |
+
"diversity_penalty": 0.0,
|
| 79 |
+
"do_sample": false,
|
| 80 |
+
"early_stopping": false,
|
| 81 |
+
"encoder_no_repeat_ngram_size": 0,
|
| 82 |
+
"eos_token_id": null,
|
| 83 |
+
"exponential_decay_length_penalty": null,
|
| 84 |
+
"finetuning_task": null,
|
| 85 |
+
"forced_bos_token_id": null,
|
| 86 |
+
"forced_eos_token_id": null,
|
| 87 |
+
"hidden_act": "silu",
|
| 88 |
+
"id2label": {
|
| 89 |
+
"0": "LABEL_0",
|
| 90 |
+
"1": "LABEL_1"
|
| 91 |
+
},
|
| 92 |
+
"is_decoder": false,
|
| 93 |
+
"is_encoder_decoder": false,
|
| 94 |
+
"label2id": {
|
| 95 |
+
"LABEL_0": 0,
|
| 96 |
+
"LABEL_1": 1
|
| 97 |
+
},
|
| 98 |
+
"length_penalty": 1.0,
|
| 99 |
+
"max_length": 20,
|
| 100 |
+
"min_length": 0,
|
| 101 |
+
"model_type": "vllama3",
|
| 102 |
+
"no_repeat_ngram_size": 0,
|
| 103 |
+
"num_beam_groups": 1,
|
| 104 |
+
"num_beams": 1,
|
| 105 |
+
"num_key_value_heads": 1,
|
| 106 |
+
"num_return_sequences": 1,
|
| 107 |
+
"output_attentions": false,
|
| 108 |
+
"output_hidden_states": false,
|
| 109 |
+
"output_scores": false,
|
| 110 |
+
"pad_token_id": null,
|
| 111 |
+
"prefix": null,
|
| 112 |
+
"problem_type": null,
|
| 113 |
+
"pruned_heads": {},
|
| 114 |
+
"qk_layer_norms_perceiver": false,
|
| 115 |
+
"remove_invalid_values": false,
|
| 116 |
+
"repetition_penalty": 1.0,
|
| 117 |
+
"resampler_depth": 6,
|
| 118 |
+
"resampler_head_dim": 96,
|
| 119 |
+
"resampler_n_heads": 16,
|
| 120 |
+
"resampler_n_latents": 64,
|
| 121 |
+
"return_dict": true,
|
| 122 |
+
"return_dict_in_generate": false,
|
| 123 |
+
"sep_token_id": null,
|
| 124 |
+
"suppress_tokens": null,
|
| 125 |
+
"task_specific_params": null,
|
| 126 |
+
"temperature": 1.0,
|
| 127 |
+
"tf_legacy_loss": false,
|
| 128 |
+
"tie_encoder_decoder": false,
|
| 129 |
+
"tie_word_embeddings": true,
|
| 130 |
+
"tokenizer_class": null,
|
| 131 |
+
"top_k": 50,
|
| 132 |
+
"top_p": 1.0,
|
| 133 |
+
"torch_dtype": null,
|
| 134 |
+
"torchscript": false,
|
| 135 |
+
"transformers_version": "4.46.0",
|
| 136 |
+
"typical_p": 1.0,
|
| 137 |
+
"use_bfloat16": false
|
| 138 |
+
},
|
| 139 |
+
"pixel_shuffle_factor": 4,
|
| 140 |
+
"pretraining_tp": 1,
|
| 141 |
+
"qk_layer_norms": false,
|
| 142 |
+
"rms_norm_eps": 1e-05,
|
| 143 |
+
"rope_interleaved": false,
|
| 144 |
+
"rope_scaling": null,
|
| 145 |
+
"rope_theta": 100000,
|
| 146 |
+
"transformers.js_config": {
|
| 147 |
+
"kv_cache_dtype": {
|
| 148 |
+
"fp16": "float16",
|
| 149 |
+
"q4f16": "float16"
|
| 150 |
+
}
|
| 151 |
+
},
|
| 152 |
+
"use_cache": true,
|
| 153 |
+
"use_resampler": false,
|
| 154 |
+
"vocab_size": 49280
|
| 155 |
+
},
|
| 156 |
+
"tie_word_embeddings": false,
|
| 157 |
+
"transformers.js_config": {
|
| 158 |
+
"kv_cache_dtype": {
|
| 159 |
+
"fp16": "float16",
|
| 160 |
+
"q4f16": "float16"
|
| 161 |
+
}
|
| 162 |
+
},
|
| 163 |
+
"use_cache": true,
|
| 164 |
+
"vision_config": {
|
| 165 |
+
"_attn_implementation_autoset": false,
|
| 166 |
+
"attention_dropout": 0.0,
|
| 167 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 168 |
+
"hidden_size": 768,
|
| 169 |
+
"image_size": 512,
|
| 170 |
+
"initializer_range": 0.02,
|
| 171 |
+
"intermediate_size": 3072,
|
| 172 |
+
"layer_norm_eps": 1e-06,
|
| 173 |
+
"max_image_size": {
|
| 174 |
+
"longest_edge": 512
|
| 175 |
+
},
|
| 176 |
+
"model_type": "idefics3_vision",
|
| 177 |
+
"num_attention_heads": 12,
|
| 178 |
+
"num_channels": 3,
|
| 179 |
+
"num_hidden_layers": 12,
|
| 180 |
+
"patch_size": 16,
|
| 181 |
+
"size": {
|
| 182 |
+
"longest_edge": 2048
|
| 183 |
+
},
|
| 184 |
+
"tie_word_embeddings": false,
|
| 185 |
+
"use_base_siglip": true
|
| 186 |
+
},
|
| 187 |
+
"vocab_size": 49280
|
| 188 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/03b0c107d1cede36875199a5d51decfe04c473de2af9999f8577a028d74d0ab4/756910ddabb85196bbca.json
ADDED
|
@@ -0,0 +1,188 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "HuggingFaceTB/SmolVLM-256M-Instruct",
|
| 4 |
+
"_task": "image-text-to-text",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Idefics3ForConditionalGeneration"
|
| 7 |
+
],
|
| 8 |
+
"dtype": "bfloat16",
|
| 9 |
+
"image_token_id": 49190,
|
| 10 |
+
"model_type": "idefics3",
|
| 11 |
+
"neuron": {
|
| 12 |
+
"_serialized_key": "NxDVLMNeuronConfig",
|
| 13 |
+
"batch_size": 1,
|
| 14 |
+
"capacity_factor": null,
|
| 15 |
+
"checkpoint_id": "HuggingFaceTB/SmolVLM-256M-Instruct",
|
| 16 |
+
"checkpoint_revision": "7e3e67edbbed1bf9888184d9df282b700a323964",
|
| 17 |
+
"continuous_batching": false,
|
| 18 |
+
"ep_degree": 1,
|
| 19 |
+
"fused_qkv": true,
|
| 20 |
+
"glu_mlp": true,
|
| 21 |
+
"image_seq_len": 64,
|
| 22 |
+
"image_size": 512,
|
| 23 |
+
"local_ranks_size": 2,
|
| 24 |
+
"max_batch_size": 1,
|
| 25 |
+
"max_context_length": 2048,
|
| 26 |
+
"max_num_images": 17,
|
| 27 |
+
"max_topk": 256,
|
| 28 |
+
"n_active_tokens": 2048,
|
| 29 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 30 |
+
"on_device_sampling": true,
|
| 31 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 32 |
+
"output_logits": false,
|
| 33 |
+
"pp_degree": 1,
|
| 34 |
+
"sequence_length": 2048,
|
| 35 |
+
"speculation_length": 0,
|
| 36 |
+
"start_rank_id": 0,
|
| 37 |
+
"target": "trn1",
|
| 38 |
+
"torch_dtype": "bfloat16",
|
| 39 |
+
"tp_degree": 2
|
| 40 |
+
},
|
| 41 |
+
"scale_factor": 4,
|
| 42 |
+
"text_config": {
|
| 43 |
+
"_attn_implementation_autoset": false,
|
| 44 |
+
"_flash_attn_2_enabled": true,
|
| 45 |
+
"_name_or_path": "None",
|
| 46 |
+
"architectures": [
|
| 47 |
+
"VLlama3ForCausalLM"
|
| 48 |
+
],
|
| 49 |
+
"attention_bias": false,
|
| 50 |
+
"attention_dropout": 0.0,
|
| 51 |
+
"dtype": "bfloat16",
|
| 52 |
+
"head_dim": 64,
|
| 53 |
+
"hidden_act": "silu",
|
| 54 |
+
"hidden_size": 576,
|
| 55 |
+
"initializer_range": 0.041666666666666664,
|
| 56 |
+
"intermediate_size": 1536,
|
| 57 |
+
"is_llama_config": true,
|
| 58 |
+
"max_position_embeddings": 8192,
|
| 59 |
+
"mlp_bias": false,
|
| 60 |
+
"model_type": "llama",
|
| 61 |
+
"neftune_noise_alpha": 0.0,
|
| 62 |
+
"num_attention_heads": 9,
|
| 63 |
+
"num_hidden_layers": 30,
|
| 64 |
+
"num_key_value_heads": 3,
|
| 65 |
+
"pad_token_id": 2,
|
| 66 |
+
"perceiver_config": {
|
| 67 |
+
"_attn_implementation_autoset": false,
|
| 68 |
+
"_name_or_path": "",
|
| 69 |
+
"add_cross_attention": false,
|
| 70 |
+
"architectures": null,
|
| 71 |
+
"attention_dropout": 0.0,
|
| 72 |
+
"bad_words_ids": null,
|
| 73 |
+
"begin_suppress_tokens": null,
|
| 74 |
+
"bos_token_id": null,
|
| 75 |
+
"chunk_size_feed_forward": 0,
|
| 76 |
+
"cross_attention_hidden_size": null,
|
| 77 |
+
"decoder_start_token_id": null,
|
| 78 |
+
"diversity_penalty": 0.0,
|
| 79 |
+
"do_sample": false,
|
| 80 |
+
"early_stopping": false,
|
| 81 |
+
"encoder_no_repeat_ngram_size": 0,
|
| 82 |
+
"eos_token_id": null,
|
| 83 |
+
"exponential_decay_length_penalty": null,
|
| 84 |
+
"finetuning_task": null,
|
| 85 |
+
"forced_bos_token_id": null,
|
| 86 |
+
"forced_eos_token_id": null,
|
| 87 |
+
"hidden_act": "silu",
|
| 88 |
+
"id2label": {
|
| 89 |
+
"0": "LABEL_0",
|
| 90 |
+
"1": "LABEL_1"
|
| 91 |
+
},
|
| 92 |
+
"is_decoder": false,
|
| 93 |
+
"is_encoder_decoder": false,
|
| 94 |
+
"label2id": {
|
| 95 |
+
"LABEL_0": 0,
|
| 96 |
+
"LABEL_1": 1
|
| 97 |
+
},
|
| 98 |
+
"length_penalty": 1.0,
|
| 99 |
+
"max_length": 20,
|
| 100 |
+
"min_length": 0,
|
| 101 |
+
"model_type": "vllama3",
|
| 102 |
+
"no_repeat_ngram_size": 0,
|
| 103 |
+
"num_beam_groups": 1,
|
| 104 |
+
"num_beams": 1,
|
| 105 |
+
"num_key_value_heads": 1,
|
| 106 |
+
"num_return_sequences": 1,
|
| 107 |
+
"output_attentions": false,
|
| 108 |
+
"output_hidden_states": false,
|
| 109 |
+
"output_scores": false,
|
| 110 |
+
"pad_token_id": null,
|
| 111 |
+
"prefix": null,
|
| 112 |
+
"problem_type": null,
|
| 113 |
+
"pruned_heads": {},
|
| 114 |
+
"qk_layer_norms_perceiver": false,
|
| 115 |
+
"remove_invalid_values": false,
|
| 116 |
+
"repetition_penalty": 1.0,
|
| 117 |
+
"resampler_depth": 6,
|
| 118 |
+
"resampler_head_dim": 96,
|
| 119 |
+
"resampler_n_heads": 16,
|
| 120 |
+
"resampler_n_latents": 64,
|
| 121 |
+
"return_dict": true,
|
| 122 |
+
"return_dict_in_generate": false,
|
| 123 |
+
"sep_token_id": null,
|
| 124 |
+
"suppress_tokens": null,
|
| 125 |
+
"task_specific_params": null,
|
| 126 |
+
"temperature": 1.0,
|
| 127 |
+
"tf_legacy_loss": false,
|
| 128 |
+
"tie_encoder_decoder": false,
|
| 129 |
+
"tie_word_embeddings": true,
|
| 130 |
+
"tokenizer_class": null,
|
| 131 |
+
"top_k": 50,
|
| 132 |
+
"top_p": 1.0,
|
| 133 |
+
"torch_dtype": null,
|
| 134 |
+
"torchscript": false,
|
| 135 |
+
"transformers_version": "4.46.0",
|
| 136 |
+
"typical_p": 1.0,
|
| 137 |
+
"use_bfloat16": false
|
| 138 |
+
},
|
| 139 |
+
"pixel_shuffle_factor": 4,
|
| 140 |
+
"pretraining_tp": 1,
|
| 141 |
+
"qk_layer_norms": false,
|
| 142 |
+
"rms_norm_eps": 1e-05,
|
| 143 |
+
"rope_interleaved": false,
|
| 144 |
+
"rope_scaling": null,
|
| 145 |
+
"rope_theta": 100000,
|
| 146 |
+
"transformers.js_config": {
|
| 147 |
+
"kv_cache_dtype": {
|
| 148 |
+
"fp16": "float16",
|
| 149 |
+
"q4f16": "float16"
|
| 150 |
+
}
|
| 151 |
+
},
|
| 152 |
+
"use_cache": true,
|
| 153 |
+
"use_resampler": false,
|
| 154 |
+
"vocab_size": 49280
|
| 155 |
+
},
|
| 156 |
+
"tie_word_embeddings": false,
|
| 157 |
+
"transformers.js_config": {
|
| 158 |
+
"kv_cache_dtype": {
|
| 159 |
+
"fp16": "float16",
|
| 160 |
+
"q4f16": "float16"
|
| 161 |
+
}
|
| 162 |
+
},
|
| 163 |
+
"use_cache": true,
|
| 164 |
+
"vision_config": {
|
| 165 |
+
"_attn_implementation_autoset": false,
|
| 166 |
+
"attention_dropout": 0.0,
|
| 167 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 168 |
+
"hidden_size": 768,
|
| 169 |
+
"image_size": 512,
|
| 170 |
+
"initializer_range": 0.02,
|
| 171 |
+
"intermediate_size": 3072,
|
| 172 |
+
"layer_norm_eps": 1e-06,
|
| 173 |
+
"max_image_size": {
|
| 174 |
+
"longest_edge": 512
|
| 175 |
+
},
|
| 176 |
+
"model_type": "idefics3_vision",
|
| 177 |
+
"num_attention_heads": 12,
|
| 178 |
+
"num_channels": 3,
|
| 179 |
+
"num_hidden_layers": 12,
|
| 180 |
+
"patch_size": 16,
|
| 181 |
+
"size": {
|
| 182 |
+
"longest_edge": 2048
|
| 183 |
+
},
|
| 184 |
+
"tie_word_embeddings": false,
|
| 185 |
+
"use_base_siglip": true
|
| 186 |
+
},
|
| 187 |
+
"vocab_size": 49280
|
| 188 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/03b0c107d1cede36875199a5d51decfe04c473de2af9999f8577a028d74d0ab4/969ba466803204052129.json
ADDED
|
@@ -0,0 +1,189 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "HuggingFaceTB/SmolVLM-256M-Instruct",
|
| 4 |
+
"_task": "image-text-to-text",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Idefics3ForConditionalGeneration"
|
| 7 |
+
],
|
| 8 |
+
"dtype": "bfloat16",
|
| 9 |
+
"image_token_id": 49190,
|
| 10 |
+
"model_type": "idefics3",
|
| 11 |
+
"neuron": {
|
| 12 |
+
"_serialized_key": "NxDVLMNeuronConfig",
|
| 13 |
+
"batch_size": 1,
|
| 14 |
+
"capacity_factor": null,
|
| 15 |
+
"checkpoint_id": "HuggingFaceTB/SmolVLM-256M-Instruct",
|
| 16 |
+
"checkpoint_revision": "7e3e67edbbed1bf9888184d9df282b700a323964",
|
| 17 |
+
"continuous_batching": false,
|
| 18 |
+
"ep_degree": 1,
|
| 19 |
+
"fused_qkv": true,
|
| 20 |
+
"glu_mlp": true,
|
| 21 |
+
"image_seq_len": 64,
|
| 22 |
+
"image_size": 512,
|
| 23 |
+
"local_ranks_size": 2,
|
| 24 |
+
"max_batch_size": 1,
|
| 25 |
+
"max_context_length": 2048,
|
| 26 |
+
"max_num_images": 17,
|
| 27 |
+
"max_topk": 256,
|
| 28 |
+
"n_active_tokens": 2048,
|
| 29 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 30 |
+
"on_device_sampling": true,
|
| 31 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 32 |
+
"output_logits": false,
|
| 33 |
+
"pp_degree": 1,
|
| 34 |
+
"prefill_chunk_size": 1024,
|
| 35 |
+
"sequence_length": 2048,
|
| 36 |
+
"speculation_length": 0,
|
| 37 |
+
"start_rank_id": 0,
|
| 38 |
+
"target": "trn1",
|
| 39 |
+
"torch_dtype": "bfloat16",
|
| 40 |
+
"tp_degree": 2
|
| 41 |
+
},
|
| 42 |
+
"scale_factor": 4,
|
| 43 |
+
"text_config": {
|
| 44 |
+
"_attn_implementation_autoset": false,
|
| 45 |
+
"_flash_attn_2_enabled": true,
|
| 46 |
+
"_name_or_path": "None",
|
| 47 |
+
"architectures": [
|
| 48 |
+
"VLlama3ForCausalLM"
|
| 49 |
+
],
|
| 50 |
+
"attention_bias": false,
|
| 51 |
+
"attention_dropout": 0.0,
|
| 52 |
+
"dtype": "bfloat16",
|
| 53 |
+
"head_dim": 64,
|
| 54 |
+
"hidden_act": "silu",
|
| 55 |
+
"hidden_size": 576,
|
| 56 |
+
"initializer_range": 0.041666666666666664,
|
| 57 |
+
"intermediate_size": 1536,
|
| 58 |
+
"is_llama_config": true,
|
| 59 |
+
"max_position_embeddings": 8192,
|
| 60 |
+
"mlp_bias": false,
|
| 61 |
+
"model_type": "llama",
|
| 62 |
+
"neftune_noise_alpha": 0.0,
|
| 63 |
+
"num_attention_heads": 9,
|
| 64 |
+
"num_hidden_layers": 30,
|
| 65 |
+
"num_key_value_heads": 3,
|
| 66 |
+
"pad_token_id": 2,
|
| 67 |
+
"perceiver_config": {
|
| 68 |
+
"_attn_implementation_autoset": false,
|
| 69 |
+
"_name_or_path": "",
|
| 70 |
+
"add_cross_attention": false,
|
| 71 |
+
"architectures": null,
|
| 72 |
+
"attention_dropout": 0.0,
|
| 73 |
+
"bad_words_ids": null,
|
| 74 |
+
"begin_suppress_tokens": null,
|
| 75 |
+
"bos_token_id": null,
|
| 76 |
+
"chunk_size_feed_forward": 0,
|
| 77 |
+
"cross_attention_hidden_size": null,
|
| 78 |
+
"decoder_start_token_id": null,
|
| 79 |
+
"diversity_penalty": 0.0,
|
| 80 |
+
"do_sample": false,
|
| 81 |
+
"early_stopping": false,
|
| 82 |
+
"encoder_no_repeat_ngram_size": 0,
|
| 83 |
+
"eos_token_id": null,
|
| 84 |
+
"exponential_decay_length_penalty": null,
|
| 85 |
+
"finetuning_task": null,
|
| 86 |
+
"forced_bos_token_id": null,
|
| 87 |
+
"forced_eos_token_id": null,
|
| 88 |
+
"hidden_act": "silu",
|
| 89 |
+
"id2label": {
|
| 90 |
+
"0": "LABEL_0",
|
| 91 |
+
"1": "LABEL_1"
|
| 92 |
+
},
|
| 93 |
+
"is_decoder": false,
|
| 94 |
+
"is_encoder_decoder": false,
|
| 95 |
+
"label2id": {
|
| 96 |
+
"LABEL_0": 0,
|
| 97 |
+
"LABEL_1": 1
|
| 98 |
+
},
|
| 99 |
+
"length_penalty": 1.0,
|
| 100 |
+
"max_length": 20,
|
| 101 |
+
"min_length": 0,
|
| 102 |
+
"model_type": "vllama3",
|
| 103 |
+
"no_repeat_ngram_size": 0,
|
| 104 |
+
"num_beam_groups": 1,
|
| 105 |
+
"num_beams": 1,
|
| 106 |
+
"num_key_value_heads": 1,
|
| 107 |
+
"num_return_sequences": 1,
|
| 108 |
+
"output_attentions": false,
|
| 109 |
+
"output_hidden_states": false,
|
| 110 |
+
"output_scores": false,
|
| 111 |
+
"pad_token_id": null,
|
| 112 |
+
"prefix": null,
|
| 113 |
+
"problem_type": null,
|
| 114 |
+
"pruned_heads": {},
|
| 115 |
+
"qk_layer_norms_perceiver": false,
|
| 116 |
+
"remove_invalid_values": false,
|
| 117 |
+
"repetition_penalty": 1.0,
|
| 118 |
+
"resampler_depth": 6,
|
| 119 |
+
"resampler_head_dim": 96,
|
| 120 |
+
"resampler_n_heads": 16,
|
| 121 |
+
"resampler_n_latents": 64,
|
| 122 |
+
"return_dict": true,
|
| 123 |
+
"return_dict_in_generate": false,
|
| 124 |
+
"sep_token_id": null,
|
| 125 |
+
"suppress_tokens": null,
|
| 126 |
+
"task_specific_params": null,
|
| 127 |
+
"temperature": 1.0,
|
| 128 |
+
"tf_legacy_loss": false,
|
| 129 |
+
"tie_encoder_decoder": false,
|
| 130 |
+
"tie_word_embeddings": true,
|
| 131 |
+
"tokenizer_class": null,
|
| 132 |
+
"top_k": 50,
|
| 133 |
+
"top_p": 1.0,
|
| 134 |
+
"torch_dtype": null,
|
| 135 |
+
"torchscript": false,
|
| 136 |
+
"transformers_version": "4.46.0",
|
| 137 |
+
"typical_p": 1.0,
|
| 138 |
+
"use_bfloat16": false
|
| 139 |
+
},
|
| 140 |
+
"pixel_shuffle_factor": 4,
|
| 141 |
+
"pretraining_tp": 1,
|
| 142 |
+
"qk_layer_norms": false,
|
| 143 |
+
"rms_norm_eps": 1e-05,
|
| 144 |
+
"rope_interleaved": false,
|
| 145 |
+
"rope_scaling": null,
|
| 146 |
+
"rope_theta": 100000,
|
| 147 |
+
"transformers.js_config": {
|
| 148 |
+
"kv_cache_dtype": {
|
| 149 |
+
"fp16": "float16",
|
| 150 |
+
"q4f16": "float16"
|
| 151 |
+
}
|
| 152 |
+
},
|
| 153 |
+
"use_cache": true,
|
| 154 |
+
"use_resampler": false,
|
| 155 |
+
"vocab_size": 49280
|
| 156 |
+
},
|
| 157 |
+
"tie_word_embeddings": false,
|
| 158 |
+
"transformers.js_config": {
|
| 159 |
+
"kv_cache_dtype": {
|
| 160 |
+
"fp16": "float16",
|
| 161 |
+
"q4f16": "float16"
|
| 162 |
+
}
|
| 163 |
+
},
|
| 164 |
+
"use_cache": true,
|
| 165 |
+
"vision_config": {
|
| 166 |
+
"_attn_implementation_autoset": false,
|
| 167 |
+
"attention_dropout": 0.0,
|
| 168 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 169 |
+
"hidden_size": 768,
|
| 170 |
+
"image_size": 512,
|
| 171 |
+
"initializer_range": 0.02,
|
| 172 |
+
"intermediate_size": 3072,
|
| 173 |
+
"layer_norm_eps": 1e-06,
|
| 174 |
+
"max_image_size": {
|
| 175 |
+
"longest_edge": 512
|
| 176 |
+
},
|
| 177 |
+
"model_type": "idefics3_vision",
|
| 178 |
+
"num_attention_heads": 12,
|
| 179 |
+
"num_channels": 3,
|
| 180 |
+
"num_hidden_layers": 12,
|
| 181 |
+
"patch_size": 16,
|
| 182 |
+
"size": {
|
| 183 |
+
"longest_edge": 2048
|
| 184 |
+
},
|
| 185 |
+
"tie_word_embeddings": false,
|
| 186 |
+
"use_base_siglip": true
|
| 187 |
+
},
|
| 188 |
+
"vocab_size": 49280
|
| 189 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/0cc25526e2cfc37a8875a3752f33c4d7505d8a07b869d0f3f41915cf6e763b74/efa7f046361aa22fb2b9.json
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "ibm-granite/granite-3.1-2b-instruct",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"GraniteForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.1,
|
| 10 |
+
"attention_multiplier": 0.015625,
|
| 11 |
+
"dtype": "bfloat16",
|
| 12 |
+
"embedding_multiplier": 12.0,
|
| 13 |
+
"hidden_act": "silu",
|
| 14 |
+
"hidden_size": 2048,
|
| 15 |
+
"initializer_range": 0.02,
|
| 16 |
+
"intermediate_size": 8192,
|
| 17 |
+
"logits_scaling": 8.0,
|
| 18 |
+
"max_position_embeddings": 131072,
|
| 19 |
+
"mlp_bias": false,
|
| 20 |
+
"model_type": "granite",
|
| 21 |
+
"neuron": {
|
| 22 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 23 |
+
"batch_size": 4,
|
| 24 |
+
"capacity_factor": null,
|
| 25 |
+
"checkpoint_id": "ibm-granite/granite-3.1-2b-instruct",
|
| 26 |
+
"checkpoint_revision": "bbc2aed595bd38bd770263dc3ab831db9794441d",
|
| 27 |
+
"continuous_batching": true,
|
| 28 |
+
"ep_degree": 1,
|
| 29 |
+
"fused_qkv": true,
|
| 30 |
+
"glu_mlp": true,
|
| 31 |
+
"local_ranks_size": 2,
|
| 32 |
+
"max_batch_size": 4,
|
| 33 |
+
"max_context_length": 4096,
|
| 34 |
+
"max_topk": 256,
|
| 35 |
+
"n_active_tokens": 4096,
|
| 36 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 37 |
+
"on_device_sampling": true,
|
| 38 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 39 |
+
"output_logits": false,
|
| 40 |
+
"pp_degree": 1,
|
| 41 |
+
"sequence_length": 4096,
|
| 42 |
+
"speculation_length": 0,
|
| 43 |
+
"start_rank_id": 0,
|
| 44 |
+
"target": "trn1",
|
| 45 |
+
"torch_dtype": "bfloat16",
|
| 46 |
+
"tp_degree": 2
|
| 47 |
+
},
|
| 48 |
+
"num_attention_heads": 32,
|
| 49 |
+
"num_hidden_layers": 40,
|
| 50 |
+
"num_key_value_heads": 8,
|
| 51 |
+
"residual_multiplier": 0.22,
|
| 52 |
+
"rms_norm_eps": 1e-05,
|
| 53 |
+
"rope_scaling": null,
|
| 54 |
+
"rope_theta": 5000000.0,
|
| 55 |
+
"tie_word_embeddings": true,
|
| 56 |
+
"use_cache": true,
|
| 57 |
+
"vocab_size": 49155
|
| 58 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/0cc25526e2cfc37a8875a3752f33c4d7505d8a07b869d0f3f41915cf6e763b74/fa7d24fcf68294ebad29.json
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "ibm-granite/granite-3.1-2b-instruct",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"GraniteForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.1,
|
| 10 |
+
"attention_multiplier": 0.015625,
|
| 11 |
+
"dtype": "bfloat16",
|
| 12 |
+
"embedding_multiplier": 12.0,
|
| 13 |
+
"hidden_act": "silu",
|
| 14 |
+
"hidden_size": 2048,
|
| 15 |
+
"initializer_range": 0.02,
|
| 16 |
+
"intermediate_size": 8192,
|
| 17 |
+
"logits_scaling": 8.0,
|
| 18 |
+
"max_position_embeddings": 131072,
|
| 19 |
+
"mlp_bias": false,
|
| 20 |
+
"model_type": "granite",
|
| 21 |
+
"neuron": {
|
| 22 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 23 |
+
"batch_size": 1,
|
| 24 |
+
"capacity_factor": null,
|
| 25 |
+
"checkpoint_id": "ibm-granite/granite-3.1-2b-instruct",
|
| 26 |
+
"checkpoint_revision": "bbc2aed595bd38bd770263dc3ab831db9794441d",
|
| 27 |
+
"continuous_batching": false,
|
| 28 |
+
"ep_degree": 1,
|
| 29 |
+
"fused_qkv": true,
|
| 30 |
+
"glu_mlp": true,
|
| 31 |
+
"local_ranks_size": 2,
|
| 32 |
+
"max_batch_size": 1,
|
| 33 |
+
"max_context_length": 8192,
|
| 34 |
+
"max_topk": 256,
|
| 35 |
+
"n_active_tokens": 8192,
|
| 36 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 37 |
+
"on_device_sampling": true,
|
| 38 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 39 |
+
"output_logits": false,
|
| 40 |
+
"pp_degree": 1,
|
| 41 |
+
"sequence_length": 8192,
|
| 42 |
+
"speculation_length": 0,
|
| 43 |
+
"start_rank_id": 0,
|
| 44 |
+
"target": "trn1",
|
| 45 |
+
"torch_dtype": "bfloat16",
|
| 46 |
+
"tp_degree": 2
|
| 47 |
+
},
|
| 48 |
+
"num_attention_heads": 32,
|
| 49 |
+
"num_hidden_layers": 40,
|
| 50 |
+
"num_key_value_heads": 8,
|
| 51 |
+
"residual_multiplier": 0.22,
|
| 52 |
+
"rms_norm_eps": 1e-05,
|
| 53 |
+
"rope_scaling": null,
|
| 54 |
+
"rope_theta": 5000000.0,
|
| 55 |
+
"tie_word_embeddings": true,
|
| 56 |
+
"use_cache": true,
|
| 57 |
+
"vocab_size": 49155
|
| 58 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/2da7a00f0478d50ae1e7f75f085c5b2773b5f355f427c61cf34cb6febd629d96/e5ddd3afb102c02baf93.json
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "yujiepan/phi-4-tiny-random",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Phi3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"auto_map": {},
|
| 11 |
+
"dtype": "bfloat16",
|
| 12 |
+
"embd_pdrop": 0.0,
|
| 13 |
+
"hidden_act": "silu",
|
| 14 |
+
"hidden_size": 16,
|
| 15 |
+
"initializer_range": 0.02,
|
| 16 |
+
"intermediate_size": 32,
|
| 17 |
+
"max_position_embeddings": 16384,
|
| 18 |
+
"model_type": "phi3",
|
| 19 |
+
"neuron": {
|
| 20 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 21 |
+
"batch_size": 1,
|
| 22 |
+
"capacity_factor": null,
|
| 23 |
+
"checkpoint_id": "yujiepan/phi-4-tiny-random",
|
| 24 |
+
"checkpoint_revision": "18a9a1168dc97ac6d128f811925670c275610f5a",
|
| 25 |
+
"continuous_batching": false,
|
| 26 |
+
"ep_degree": 1,
|
| 27 |
+
"fused_qkv": true,
|
| 28 |
+
"glu_mlp": true,
|
| 29 |
+
"local_ranks_size": 2,
|
| 30 |
+
"max_batch_size": 1,
|
| 31 |
+
"max_context_length": 1024,
|
| 32 |
+
"max_topk": 256,
|
| 33 |
+
"n_active_tokens": 1024,
|
| 34 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 35 |
+
"on_device_sampling": true,
|
| 36 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 37 |
+
"output_logits": false,
|
| 38 |
+
"pp_degree": 1,
|
| 39 |
+
"sequence_length": 1024,
|
| 40 |
+
"speculation_length": 0,
|
| 41 |
+
"start_rank_id": 0,
|
| 42 |
+
"target": "trn1",
|
| 43 |
+
"torch_dtype": "bfloat16",
|
| 44 |
+
"tp_degree": 2
|
| 45 |
+
},
|
| 46 |
+
"num_attention_heads": 2,
|
| 47 |
+
"num_hidden_layers": 2,
|
| 48 |
+
"num_key_value_heads": 1,
|
| 49 |
+
"original_max_position_embeddings": 16384,
|
| 50 |
+
"partial_rotary_factor": 1.0,
|
| 51 |
+
"resid_pdrop": 0.0,
|
| 52 |
+
"rms_norm_eps": 1e-05,
|
| 53 |
+
"rope_scaling": null,
|
| 54 |
+
"rope_theta": 250000,
|
| 55 |
+
"sliding_window": null,
|
| 56 |
+
"tie_word_embeddings": false,
|
| 57 |
+
"use_cache": true,
|
| 58 |
+
"vocab_size": 100352
|
| 59 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/441269935591cad8d370e512c0b93cdd2fce6247c40e5a4866d872ee5338b0de/51c77185b9832eaebdfc.json
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "unsloth/gemma-3-270m-it",
|
| 4 |
+
"_sliding_window_pattern": 6,
|
| 5 |
+
"_task": "text-generation",
|
| 6 |
+
"architectures": [
|
| 7 |
+
"Gemma3ForCausalLM"
|
| 8 |
+
],
|
| 9 |
+
"attention_bias": false,
|
| 10 |
+
"attention_dropout": 0.0,
|
| 11 |
+
"attn_logit_softcapping": null,
|
| 12 |
+
"dtype": "bfloat16",
|
| 13 |
+
"final_logit_softcapping": null,
|
| 14 |
+
"head_dim": 256,
|
| 15 |
+
"hidden_activation": "gelu_pytorch_tanh",
|
| 16 |
+
"hidden_size": 640,
|
| 17 |
+
"initializer_range": 0.02,
|
| 18 |
+
"intermediate_size": 2048,
|
| 19 |
+
"layer_types": [
|
| 20 |
+
"sliding_attention",
|
| 21 |
+
"sliding_attention",
|
| 22 |
+
"sliding_attention",
|
| 23 |
+
"sliding_attention",
|
| 24 |
+
"sliding_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"sliding_attention",
|
| 27 |
+
"sliding_attention",
|
| 28 |
+
"sliding_attention",
|
| 29 |
+
"sliding_attention",
|
| 30 |
+
"sliding_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"sliding_attention",
|
| 33 |
+
"sliding_attention",
|
| 34 |
+
"sliding_attention",
|
| 35 |
+
"sliding_attention",
|
| 36 |
+
"sliding_attention",
|
| 37 |
+
"full_attention"
|
| 38 |
+
],
|
| 39 |
+
"max_position_embeddings": 32768,
|
| 40 |
+
"model_type": "gemma3_text",
|
| 41 |
+
"neuron": {
|
| 42 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 43 |
+
"batch_size": 1,
|
| 44 |
+
"capacity_factor": null,
|
| 45 |
+
"checkpoint_id": "unsloth/gemma-3-270m-it",
|
| 46 |
+
"checkpoint_revision": "23cf460f6bb16954176b3ddcc8d4f250501458a9",
|
| 47 |
+
"continuous_batching": false,
|
| 48 |
+
"ep_degree": 1,
|
| 49 |
+
"fused_qkv": true,
|
| 50 |
+
"glu_mlp": true,
|
| 51 |
+
"local_ranks_size": 2,
|
| 52 |
+
"max_batch_size": 1,
|
| 53 |
+
"max_context_length": 8192,
|
| 54 |
+
"max_topk": 256,
|
| 55 |
+
"n_active_tokens": 8192,
|
| 56 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 57 |
+
"on_device_sampling": true,
|
| 58 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 59 |
+
"output_logits": false,
|
| 60 |
+
"pp_degree": 1,
|
| 61 |
+
"sequence_length": 8192,
|
| 62 |
+
"speculation_length": 0,
|
| 63 |
+
"start_rank_id": 0,
|
| 64 |
+
"target": "trn1",
|
| 65 |
+
"torch_dtype": "bfloat16",
|
| 66 |
+
"tp_degree": 2
|
| 67 |
+
},
|
| 68 |
+
"num_attention_heads": 4,
|
| 69 |
+
"num_hidden_layers": 18,
|
| 70 |
+
"num_key_value_heads": 1,
|
| 71 |
+
"query_pre_attn_scalar": 256,
|
| 72 |
+
"rms_norm_eps": 1e-06,
|
| 73 |
+
"rope_local_base_freq": 10000.0,
|
| 74 |
+
"rope_scaling": null,
|
| 75 |
+
"rope_theta": 1000000.0,
|
| 76 |
+
"sliding_window": 512,
|
| 77 |
+
"unsloth_fixed": true,
|
| 78 |
+
"use_bidirectional_attention": false,
|
| 79 |
+
"use_cache": true,
|
| 80 |
+
"vocab_size": 262144
|
| 81 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/441269935591cad8d370e512c0b93cdd2fce6247c40e5a4866d872ee5338b0de/51eb5f08405da966cefd.json
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "unsloth/gemma-3-270m-it",
|
| 4 |
+
"_sliding_window_pattern": 6,
|
| 5 |
+
"_task": "text-generation",
|
| 6 |
+
"architectures": [
|
| 7 |
+
"Gemma3ForCausalLM"
|
| 8 |
+
],
|
| 9 |
+
"attention_bias": false,
|
| 10 |
+
"attention_dropout": 0.0,
|
| 11 |
+
"attn_logit_softcapping": null,
|
| 12 |
+
"dtype": "bfloat16",
|
| 13 |
+
"final_logit_softcapping": null,
|
| 14 |
+
"head_dim": 256,
|
| 15 |
+
"hidden_activation": "gelu_pytorch_tanh",
|
| 16 |
+
"hidden_size": 640,
|
| 17 |
+
"initializer_range": 0.02,
|
| 18 |
+
"intermediate_size": 2048,
|
| 19 |
+
"layer_types": [
|
| 20 |
+
"sliding_attention",
|
| 21 |
+
"sliding_attention",
|
| 22 |
+
"sliding_attention",
|
| 23 |
+
"sliding_attention",
|
| 24 |
+
"sliding_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"sliding_attention",
|
| 27 |
+
"sliding_attention",
|
| 28 |
+
"sliding_attention",
|
| 29 |
+
"sliding_attention",
|
| 30 |
+
"sliding_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"sliding_attention",
|
| 33 |
+
"sliding_attention",
|
| 34 |
+
"sliding_attention",
|
| 35 |
+
"sliding_attention",
|
| 36 |
+
"sliding_attention",
|
| 37 |
+
"full_attention"
|
| 38 |
+
],
|
| 39 |
+
"max_position_embeddings": 32768,
|
| 40 |
+
"model_type": "gemma3_text",
|
| 41 |
+
"neuron": {
|
| 42 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 43 |
+
"batch_size": 1,
|
| 44 |
+
"capacity_factor": null,
|
| 45 |
+
"checkpoint_id": "unsloth/gemma-3-270m-it",
|
| 46 |
+
"checkpoint_revision": "23cf460f6bb16954176b3ddcc8d4f250501458a9",
|
| 47 |
+
"continuous_batching": false,
|
| 48 |
+
"ep_degree": 1,
|
| 49 |
+
"fused_qkv": true,
|
| 50 |
+
"glu_mlp": true,
|
| 51 |
+
"local_ranks_size": 2,
|
| 52 |
+
"max_batch_size": 1,
|
| 53 |
+
"max_context_length": 1024,
|
| 54 |
+
"max_topk": 256,
|
| 55 |
+
"n_active_tokens": 1024,
|
| 56 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 57 |
+
"on_device_sampling": true,
|
| 58 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 59 |
+
"output_logits": false,
|
| 60 |
+
"pp_degree": 1,
|
| 61 |
+
"sequence_length": 1024,
|
| 62 |
+
"speculation_length": 0,
|
| 63 |
+
"start_rank_id": 0,
|
| 64 |
+
"target": "trn1",
|
| 65 |
+
"torch_dtype": "bfloat16",
|
| 66 |
+
"tp_degree": 2
|
| 67 |
+
},
|
| 68 |
+
"num_attention_heads": 4,
|
| 69 |
+
"num_hidden_layers": 18,
|
| 70 |
+
"num_key_value_heads": 1,
|
| 71 |
+
"query_pre_attn_scalar": 256,
|
| 72 |
+
"rms_norm_eps": 1e-06,
|
| 73 |
+
"rope_local_base_freq": 10000.0,
|
| 74 |
+
"rope_scaling": null,
|
| 75 |
+
"rope_theta": 1000000.0,
|
| 76 |
+
"sliding_window": 512,
|
| 77 |
+
"unsloth_fixed": true,
|
| 78 |
+
"use_bidirectional_attention": false,
|
| 79 |
+
"use_cache": true,
|
| 80 |
+
"vocab_size": 262144
|
| 81 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/441269935591cad8d370e512c0b93cdd2fce6247c40e5a4866d872ee5338b0de/6109da25218b5116e9b5.json
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "unsloth/gemma-3-270m-it",
|
| 4 |
+
"_sliding_window_pattern": 6,
|
| 5 |
+
"_task": "text-generation",
|
| 6 |
+
"architectures": [
|
| 7 |
+
"Gemma3ForCausalLM"
|
| 8 |
+
],
|
| 9 |
+
"attention_bias": false,
|
| 10 |
+
"attention_dropout": 0.0,
|
| 11 |
+
"attn_logit_softcapping": null,
|
| 12 |
+
"dtype": "bfloat16",
|
| 13 |
+
"final_logit_softcapping": null,
|
| 14 |
+
"head_dim": 256,
|
| 15 |
+
"hidden_activation": "gelu_pytorch_tanh",
|
| 16 |
+
"hidden_size": 640,
|
| 17 |
+
"initializer_range": 0.02,
|
| 18 |
+
"intermediate_size": 2048,
|
| 19 |
+
"layer_types": [
|
| 20 |
+
"sliding_attention",
|
| 21 |
+
"sliding_attention",
|
| 22 |
+
"sliding_attention",
|
| 23 |
+
"sliding_attention",
|
| 24 |
+
"sliding_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"sliding_attention",
|
| 27 |
+
"sliding_attention",
|
| 28 |
+
"sliding_attention",
|
| 29 |
+
"sliding_attention",
|
| 30 |
+
"sliding_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"sliding_attention",
|
| 33 |
+
"sliding_attention",
|
| 34 |
+
"sliding_attention",
|
| 35 |
+
"sliding_attention",
|
| 36 |
+
"sliding_attention",
|
| 37 |
+
"full_attention"
|
| 38 |
+
],
|
| 39 |
+
"max_position_embeddings": 32768,
|
| 40 |
+
"model_type": "gemma3_text",
|
| 41 |
+
"neuron": {
|
| 42 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 43 |
+
"batch_size": 4,
|
| 44 |
+
"capacity_factor": null,
|
| 45 |
+
"checkpoint_id": "unsloth/gemma-3-270m-it",
|
| 46 |
+
"checkpoint_revision": "23cf460f6bb16954176b3ddcc8d4f250501458a9",
|
| 47 |
+
"continuous_batching": true,
|
| 48 |
+
"ep_degree": 1,
|
| 49 |
+
"fused_qkv": true,
|
| 50 |
+
"glu_mlp": true,
|
| 51 |
+
"local_ranks_size": 2,
|
| 52 |
+
"max_batch_size": 4,
|
| 53 |
+
"max_context_length": 4096,
|
| 54 |
+
"max_topk": 256,
|
| 55 |
+
"n_active_tokens": 4096,
|
| 56 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 57 |
+
"on_device_sampling": true,
|
| 58 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 59 |
+
"output_logits": false,
|
| 60 |
+
"pp_degree": 1,
|
| 61 |
+
"sequence_length": 4096,
|
| 62 |
+
"speculation_length": 0,
|
| 63 |
+
"start_rank_id": 0,
|
| 64 |
+
"target": "trn1",
|
| 65 |
+
"torch_dtype": "bfloat16",
|
| 66 |
+
"tp_degree": 2
|
| 67 |
+
},
|
| 68 |
+
"num_attention_heads": 4,
|
| 69 |
+
"num_hidden_layers": 18,
|
| 70 |
+
"num_key_value_heads": 1,
|
| 71 |
+
"query_pre_attn_scalar": 256,
|
| 72 |
+
"rms_norm_eps": 1e-06,
|
| 73 |
+
"rope_local_base_freq": 10000.0,
|
| 74 |
+
"rope_scaling": null,
|
| 75 |
+
"rope_theta": 1000000.0,
|
| 76 |
+
"sliding_window": 512,
|
| 77 |
+
"unsloth_fixed": true,
|
| 78 |
+
"use_bidirectional_attention": false,
|
| 79 |
+
"use_cache": true,
|
| 80 |
+
"vocab_size": 262144
|
| 81 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/4ab8140bc7eb4a553d95855c5c2be2cf8c0fbab21b823d76183b6f51e98b6fc5/03a64a22d1b885eece61.json
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "dacorvo/Mixtral-tiny",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"MixtralForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_dropout": 0.0,
|
| 9 |
+
"dtype": "float16",
|
| 10 |
+
"head_dim": 32,
|
| 11 |
+
"hidden_act": "silu",
|
| 12 |
+
"hidden_size": 1024,
|
| 13 |
+
"initializer_range": 0.02,
|
| 14 |
+
"intermediate_size": 3584,
|
| 15 |
+
"max_position_embeddings": 1024,
|
| 16 |
+
"model_type": "mixtral",
|
| 17 |
+
"neuron": {
|
| 18 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 19 |
+
"batch_size": 1,
|
| 20 |
+
"capacity_factor": null,
|
| 21 |
+
"checkpoint_id": "dacorvo/Mixtral-tiny",
|
| 22 |
+
"checkpoint_revision": "c557ba205ddff6ea911f4719e0d543d6c08356b6",
|
| 23 |
+
"continuous_batching": false,
|
| 24 |
+
"ep_degree": 1,
|
| 25 |
+
"fused_qkv": false,
|
| 26 |
+
"glu_mlp": true,
|
| 27 |
+
"local_ranks_size": 2,
|
| 28 |
+
"max_batch_size": 1,
|
| 29 |
+
"max_context_length": 1024,
|
| 30 |
+
"max_topk": 256,
|
| 31 |
+
"n_active_tokens": 1024,
|
| 32 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 33 |
+
"on_device_sampling": false,
|
| 34 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 35 |
+
"output_logits": false,
|
| 36 |
+
"pp_degree": 1,
|
| 37 |
+
"sequence_length": 1024,
|
| 38 |
+
"speculation_length": 0,
|
| 39 |
+
"start_rank_id": 0,
|
| 40 |
+
"target": "trn1",
|
| 41 |
+
"torch_dtype": "float16",
|
| 42 |
+
"tp_degree": 2
|
| 43 |
+
},
|
| 44 |
+
"num_attention_heads": 32,
|
| 45 |
+
"num_experts_per_tok": 2,
|
| 46 |
+
"num_hidden_layers": 2,
|
| 47 |
+
"num_key_value_heads": 8,
|
| 48 |
+
"num_local_experts": 8,
|
| 49 |
+
"output_router_logits": false,
|
| 50 |
+
"rms_norm_eps": 1e-05,
|
| 51 |
+
"rope_theta": 10000.0,
|
| 52 |
+
"router_aux_loss_coef": 0.001,
|
| 53 |
+
"router_jitter_noise": 0.0,
|
| 54 |
+
"sliding_window": 4096,
|
| 55 |
+
"tie_word_embeddings": false,
|
| 56 |
+
"use_cache": true,
|
| 57 |
+
"vocab_size": 32000
|
| 58 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/4cb7aff9e2a15c151396f2b684013e39d6739f0dec83e5c9dabbfe9d5fcf77b7/1c8b4a21eb41ff235945.json
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen2.5-0.5B",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen2ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_dropout": 0.0,
|
| 9 |
+
"dtype": "bfloat16",
|
| 10 |
+
"hidden_act": "silu",
|
| 11 |
+
"hidden_size": 896,
|
| 12 |
+
"initializer_range": 0.02,
|
| 13 |
+
"intermediate_size": 4864,
|
| 14 |
+
"layer_types": [
|
| 15 |
+
"full_attention",
|
| 16 |
+
"full_attention",
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention"
|
| 39 |
+
],
|
| 40 |
+
"max_position_embeddings": 32768,
|
| 41 |
+
"max_window_layers": 24,
|
| 42 |
+
"model_type": "qwen2",
|
| 43 |
+
"neuron": {
|
| 44 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 45 |
+
"batch_size": 4,
|
| 46 |
+
"capacity_factor": null,
|
| 47 |
+
"checkpoint_id": "Qwen/Qwen2.5-0.5B",
|
| 48 |
+
"checkpoint_revision": "060db6499f32faf8b98477b0a26969ef7d8b9987",
|
| 49 |
+
"continuous_batching": true,
|
| 50 |
+
"ep_degree": 1,
|
| 51 |
+
"fused_qkv": false,
|
| 52 |
+
"glu_mlp": true,
|
| 53 |
+
"local_ranks_size": 2,
|
| 54 |
+
"max_batch_size": 4,
|
| 55 |
+
"max_context_length": 4096,
|
| 56 |
+
"max_topk": 256,
|
| 57 |
+
"n_active_tokens": 4096,
|
| 58 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 59 |
+
"on_device_sampling": false,
|
| 60 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 61 |
+
"output_logits": false,
|
| 62 |
+
"pp_degree": 1,
|
| 63 |
+
"sequence_length": 4096,
|
| 64 |
+
"speculation_length": 0,
|
| 65 |
+
"start_rank_id": 0,
|
| 66 |
+
"target": "trn1",
|
| 67 |
+
"torch_dtype": "bfloat16",
|
| 68 |
+
"tp_degree": 2
|
| 69 |
+
},
|
| 70 |
+
"num_attention_heads": 14,
|
| 71 |
+
"num_hidden_layers": 24,
|
| 72 |
+
"num_key_value_heads": 2,
|
| 73 |
+
"rms_norm_eps": 1e-06,
|
| 74 |
+
"rope_scaling": null,
|
| 75 |
+
"rope_theta": 1000000.0,
|
| 76 |
+
"sliding_window": null,
|
| 77 |
+
"tie_word_embeddings": true,
|
| 78 |
+
"use_cache": true,
|
| 79 |
+
"use_mrope": false,
|
| 80 |
+
"use_sliding_window": false,
|
| 81 |
+
"vocab_size": 151936
|
| 82 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/4cb7aff9e2a15c151396f2b684013e39d6739f0dec83e5c9dabbfe9d5fcf77b7/b12b8be52c487dcc560f.json
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen2.5-0.5B",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen2ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_dropout": 0.0,
|
| 9 |
+
"dtype": "bfloat16",
|
| 10 |
+
"hidden_act": "silu",
|
| 11 |
+
"hidden_size": 896,
|
| 12 |
+
"initializer_range": 0.02,
|
| 13 |
+
"intermediate_size": 4864,
|
| 14 |
+
"layer_types": [
|
| 15 |
+
"full_attention",
|
| 16 |
+
"full_attention",
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention"
|
| 39 |
+
],
|
| 40 |
+
"max_position_embeddings": 32768,
|
| 41 |
+
"max_window_layers": 24,
|
| 42 |
+
"model_type": "qwen2",
|
| 43 |
+
"neuron": {
|
| 44 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 45 |
+
"batch_size": 1,
|
| 46 |
+
"capacity_factor": null,
|
| 47 |
+
"checkpoint_id": "Qwen/Qwen2.5-0.5B",
|
| 48 |
+
"checkpoint_revision": "060db6499f32faf8b98477b0a26969ef7d8b9987",
|
| 49 |
+
"continuous_batching": false,
|
| 50 |
+
"ep_degree": 1,
|
| 51 |
+
"fused_qkv": false,
|
| 52 |
+
"glu_mlp": true,
|
| 53 |
+
"local_ranks_size": 2,
|
| 54 |
+
"max_batch_size": 1,
|
| 55 |
+
"max_context_length": 8192,
|
| 56 |
+
"max_topk": 256,
|
| 57 |
+
"n_active_tokens": 8192,
|
| 58 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 59 |
+
"on_device_sampling": true,
|
| 60 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 61 |
+
"output_logits": false,
|
| 62 |
+
"pp_degree": 1,
|
| 63 |
+
"sequence_length": 8192,
|
| 64 |
+
"speculation_length": 0,
|
| 65 |
+
"start_rank_id": 0,
|
| 66 |
+
"target": "trn1",
|
| 67 |
+
"torch_dtype": "bfloat16",
|
| 68 |
+
"tp_degree": 2
|
| 69 |
+
},
|
| 70 |
+
"num_attention_heads": 14,
|
| 71 |
+
"num_hidden_layers": 24,
|
| 72 |
+
"num_key_value_heads": 2,
|
| 73 |
+
"rms_norm_eps": 1e-06,
|
| 74 |
+
"rope_scaling": null,
|
| 75 |
+
"rope_theta": 1000000.0,
|
| 76 |
+
"sliding_window": null,
|
| 77 |
+
"tie_word_embeddings": true,
|
| 78 |
+
"use_cache": true,
|
| 79 |
+
"use_mrope": false,
|
| 80 |
+
"use_sliding_window": false,
|
| 81 |
+
"vocab_size": 151936
|
| 82 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/6454afdf3e9d66c7226c13a575b718845c25e53b0699600ba2bb4f883e9d841b/97e03fcfc7f46ac8836e.json
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "llamafactory/tiny-random-Llama-3",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"LlamaForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "float16",
|
| 11 |
+
"head_dim": 4,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 16,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 64,
|
| 16 |
+
"max_position_embeddings": 131072,
|
| 17 |
+
"mlp_bias": false,
|
| 18 |
+
"model_type": "llama",
|
| 19 |
+
"neuron": {
|
| 20 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 21 |
+
"batch_size": 1,
|
| 22 |
+
"capacity_factor": null,
|
| 23 |
+
"checkpoint_id": "llamafactory/tiny-random-Llama-3",
|
| 24 |
+
"checkpoint_revision": "bf2a2e3bf199ad2ee96f02a3c00246c608db22a8",
|
| 25 |
+
"continuous_batching": false,
|
| 26 |
+
"ep_degree": 1,
|
| 27 |
+
"fused_qkv": true,
|
| 28 |
+
"glu_mlp": true,
|
| 29 |
+
"local_ranks_size": 2,
|
| 30 |
+
"max_batch_size": 1,
|
| 31 |
+
"max_context_length": 1024,
|
| 32 |
+
"max_topk": 256,
|
| 33 |
+
"n_active_tokens": 1024,
|
| 34 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 35 |
+
"on_device_sampling": true,
|
| 36 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 37 |
+
"output_logits": false,
|
| 38 |
+
"pp_degree": 1,
|
| 39 |
+
"sequence_length": 1024,
|
| 40 |
+
"speculation_length": 0,
|
| 41 |
+
"start_rank_id": 0,
|
| 42 |
+
"target": "trn1",
|
| 43 |
+
"torch_dtype": "float16",
|
| 44 |
+
"tp_degree": 2
|
| 45 |
+
},
|
| 46 |
+
"num_attention_heads": 4,
|
| 47 |
+
"num_hidden_layers": 2,
|
| 48 |
+
"num_key_value_heads": 4,
|
| 49 |
+
"pretraining_tp": 1,
|
| 50 |
+
"rms_norm_eps": 1e-05,
|
| 51 |
+
"rope_scaling": {
|
| 52 |
+
"factor": 8.0,
|
| 53 |
+
"high_freq_factor": 4.0,
|
| 54 |
+
"low_freq_factor": 1.0,
|
| 55 |
+
"original_max_position_embeddings": 8192,
|
| 56 |
+
"rope_type": "llama3"
|
| 57 |
+
},
|
| 58 |
+
"rope_theta": 500000.0,
|
| 59 |
+
"tie_word_embeddings": false,
|
| 60 |
+
"use_cache": true,
|
| 61 |
+
"vocab_size": 128256
|
| 62 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/73707b485eab9008c7aba7f5dad0ce2384ac685318d5f888c12fa0d81ed90b19/078d41a850db4b8221d6.json
ADDED
|
@@ -0,0 +1,134 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "HuggingFaceTB/SmolLM3-3B",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"SmolLM3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"hidden_act": "silu",
|
| 12 |
+
"hidden_size": 2048,
|
| 13 |
+
"initializer_range": 0.02,
|
| 14 |
+
"intermediate_size": 11008,
|
| 15 |
+
"layer_types": [
|
| 16 |
+
"full_attention",
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention"
|
| 52 |
+
],
|
| 53 |
+
"max_position_embeddings": 65536,
|
| 54 |
+
"max_window_layers": 28,
|
| 55 |
+
"mlp_bias": false,
|
| 56 |
+
"model_type": "smollm3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 1,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "HuggingFaceTB/SmolLM3-3B",
|
| 62 |
+
"checkpoint_revision": "a07cc9a04f16550a088caea529712d1d335b0ac1",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 1,
|
| 69 |
+
"max_context_length": 8192,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 8192,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": true,
|
| 74 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 8192,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"no_rope_layer_interval": 4,
|
| 85 |
+
"no_rope_layers": [
|
| 86 |
+
1,
|
| 87 |
+
1,
|
| 88 |
+
1,
|
| 89 |
+
0,
|
| 90 |
+
1,
|
| 91 |
+
1,
|
| 92 |
+
1,
|
| 93 |
+
0,
|
| 94 |
+
1,
|
| 95 |
+
1,
|
| 96 |
+
1,
|
| 97 |
+
0,
|
| 98 |
+
1,
|
| 99 |
+
1,
|
| 100 |
+
1,
|
| 101 |
+
0,
|
| 102 |
+
1,
|
| 103 |
+
1,
|
| 104 |
+
1,
|
| 105 |
+
0,
|
| 106 |
+
1,
|
| 107 |
+
1,
|
| 108 |
+
1,
|
| 109 |
+
0,
|
| 110 |
+
1,
|
| 111 |
+
1,
|
| 112 |
+
1,
|
| 113 |
+
0,
|
| 114 |
+
1,
|
| 115 |
+
1,
|
| 116 |
+
1,
|
| 117 |
+
0,
|
| 118 |
+
1,
|
| 119 |
+
1,
|
| 120 |
+
1,
|
| 121 |
+
0
|
| 122 |
+
],
|
| 123 |
+
"num_attention_heads": 16,
|
| 124 |
+
"num_hidden_layers": 36,
|
| 125 |
+
"num_key_value_heads": 4,
|
| 126 |
+
"pretraining_tp": 2,
|
| 127 |
+
"rms_norm_eps": 1e-06,
|
| 128 |
+
"rope_scaling": null,
|
| 129 |
+
"rope_theta": 5000000.0,
|
| 130 |
+
"sliding_window": null,
|
| 131 |
+
"use_cache": false,
|
| 132 |
+
"use_sliding_window": false,
|
| 133 |
+
"vocab_size": 128256
|
| 134 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/73707b485eab9008c7aba7f5dad0ce2384ac685318d5f888c12fa0d81ed90b19/2dc771f3c35b9af34ce4.json
ADDED
|
@@ -0,0 +1,134 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "HuggingFaceTB/SmolLM3-3B",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"SmolLM3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"hidden_act": "silu",
|
| 12 |
+
"hidden_size": 2048,
|
| 13 |
+
"initializer_range": 0.02,
|
| 14 |
+
"intermediate_size": 11008,
|
| 15 |
+
"layer_types": [
|
| 16 |
+
"full_attention",
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention"
|
| 52 |
+
],
|
| 53 |
+
"max_position_embeddings": 65536,
|
| 54 |
+
"max_window_layers": 28,
|
| 55 |
+
"mlp_bias": false,
|
| 56 |
+
"model_type": "smollm3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 4,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "HuggingFaceTB/SmolLM3-3B",
|
| 62 |
+
"checkpoint_revision": "a07cc9a04f16550a088caea529712d1d335b0ac1",
|
| 63 |
+
"continuous_batching": true,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 4,
|
| 69 |
+
"max_context_length": 4096,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 4096,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": true,
|
| 74 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 4096,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"no_rope_layer_interval": 4,
|
| 85 |
+
"no_rope_layers": [
|
| 86 |
+
1,
|
| 87 |
+
1,
|
| 88 |
+
1,
|
| 89 |
+
0,
|
| 90 |
+
1,
|
| 91 |
+
1,
|
| 92 |
+
1,
|
| 93 |
+
0,
|
| 94 |
+
1,
|
| 95 |
+
1,
|
| 96 |
+
1,
|
| 97 |
+
0,
|
| 98 |
+
1,
|
| 99 |
+
1,
|
| 100 |
+
1,
|
| 101 |
+
0,
|
| 102 |
+
1,
|
| 103 |
+
1,
|
| 104 |
+
1,
|
| 105 |
+
0,
|
| 106 |
+
1,
|
| 107 |
+
1,
|
| 108 |
+
1,
|
| 109 |
+
0,
|
| 110 |
+
1,
|
| 111 |
+
1,
|
| 112 |
+
1,
|
| 113 |
+
0,
|
| 114 |
+
1,
|
| 115 |
+
1,
|
| 116 |
+
1,
|
| 117 |
+
0,
|
| 118 |
+
1,
|
| 119 |
+
1,
|
| 120 |
+
1,
|
| 121 |
+
0
|
| 122 |
+
],
|
| 123 |
+
"num_attention_heads": 16,
|
| 124 |
+
"num_hidden_layers": 36,
|
| 125 |
+
"num_key_value_heads": 4,
|
| 126 |
+
"pretraining_tp": 2,
|
| 127 |
+
"rms_norm_eps": 1e-06,
|
| 128 |
+
"rope_scaling": null,
|
| 129 |
+
"rope_theta": 5000000.0,
|
| 130 |
+
"sliding_window": null,
|
| 131 |
+
"use_cache": false,
|
| 132 |
+
"use_sliding_window": false,
|
| 133 |
+
"vocab_size": 128256
|
| 134 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/73707b485eab9008c7aba7f5dad0ce2384ac685318d5f888c12fa0d81ed90b19/d04c2a3f18746af5f901.json
ADDED
|
@@ -0,0 +1,134 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "HuggingFaceTB/SmolLM3-3B",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"SmolLM3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"hidden_act": "silu",
|
| 12 |
+
"hidden_size": 2048,
|
| 13 |
+
"initializer_range": 0.02,
|
| 14 |
+
"intermediate_size": 11008,
|
| 15 |
+
"layer_types": [
|
| 16 |
+
"full_attention",
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention",
|
| 45 |
+
"full_attention",
|
| 46 |
+
"full_attention",
|
| 47 |
+
"full_attention",
|
| 48 |
+
"full_attention",
|
| 49 |
+
"full_attention",
|
| 50 |
+
"full_attention",
|
| 51 |
+
"full_attention"
|
| 52 |
+
],
|
| 53 |
+
"max_position_embeddings": 65536,
|
| 54 |
+
"max_window_layers": 28,
|
| 55 |
+
"mlp_bias": false,
|
| 56 |
+
"model_type": "smollm3",
|
| 57 |
+
"neuron": {
|
| 58 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 59 |
+
"batch_size": 1,
|
| 60 |
+
"capacity_factor": null,
|
| 61 |
+
"checkpoint_id": "HuggingFaceTB/SmolLM3-3B",
|
| 62 |
+
"checkpoint_revision": "a07cc9a04f16550a088caea529712d1d335b0ac1",
|
| 63 |
+
"continuous_batching": false,
|
| 64 |
+
"ep_degree": 1,
|
| 65 |
+
"fused_qkv": true,
|
| 66 |
+
"glu_mlp": true,
|
| 67 |
+
"local_ranks_size": 2,
|
| 68 |
+
"max_batch_size": 1,
|
| 69 |
+
"max_context_length": 1024,
|
| 70 |
+
"max_topk": 256,
|
| 71 |
+
"n_active_tokens": 1024,
|
| 72 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 73 |
+
"on_device_sampling": true,
|
| 74 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 75 |
+
"output_logits": false,
|
| 76 |
+
"pp_degree": 1,
|
| 77 |
+
"sequence_length": 1024,
|
| 78 |
+
"speculation_length": 0,
|
| 79 |
+
"start_rank_id": 0,
|
| 80 |
+
"target": "trn1",
|
| 81 |
+
"torch_dtype": "bfloat16",
|
| 82 |
+
"tp_degree": 2
|
| 83 |
+
},
|
| 84 |
+
"no_rope_layer_interval": 4,
|
| 85 |
+
"no_rope_layers": [
|
| 86 |
+
1,
|
| 87 |
+
1,
|
| 88 |
+
1,
|
| 89 |
+
0,
|
| 90 |
+
1,
|
| 91 |
+
1,
|
| 92 |
+
1,
|
| 93 |
+
0,
|
| 94 |
+
1,
|
| 95 |
+
1,
|
| 96 |
+
1,
|
| 97 |
+
0,
|
| 98 |
+
1,
|
| 99 |
+
1,
|
| 100 |
+
1,
|
| 101 |
+
0,
|
| 102 |
+
1,
|
| 103 |
+
1,
|
| 104 |
+
1,
|
| 105 |
+
0,
|
| 106 |
+
1,
|
| 107 |
+
1,
|
| 108 |
+
1,
|
| 109 |
+
0,
|
| 110 |
+
1,
|
| 111 |
+
1,
|
| 112 |
+
1,
|
| 113 |
+
0,
|
| 114 |
+
1,
|
| 115 |
+
1,
|
| 116 |
+
1,
|
| 117 |
+
0,
|
| 118 |
+
1,
|
| 119 |
+
1,
|
| 120 |
+
1,
|
| 121 |
+
0
|
| 122 |
+
],
|
| 123 |
+
"num_attention_heads": 16,
|
| 124 |
+
"num_hidden_layers": 36,
|
| 125 |
+
"num_key_value_heads": 4,
|
| 126 |
+
"pretraining_tp": 2,
|
| 127 |
+
"rms_norm_eps": 1e-06,
|
| 128 |
+
"rope_scaling": null,
|
| 129 |
+
"rope_theta": 5000000.0,
|
| 130 |
+
"sliding_window": null,
|
| 131 |
+
"use_cache": false,
|
| 132 |
+
"use_sliding_window": false,
|
| 133 |
+
"vocab_size": 128256
|
| 134 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/7518518c7e077820070186deda960d8cc49db068cdf0ac70664098fa2b6b698c/29c61ad2f54baaec4c1d.json
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "yujiepan/qwen2.5-128k-tiny-random",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen2ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_dropout": 0.0,
|
| 9 |
+
"dtype": "bfloat16",
|
| 10 |
+
"hidden_act": "silu",
|
| 11 |
+
"hidden_size": 8,
|
| 12 |
+
"initializer_range": 0.02,
|
| 13 |
+
"intermediate_size": 16,
|
| 14 |
+
"layer_types": [
|
| 15 |
+
"full_attention",
|
| 16 |
+
"full_attention"
|
| 17 |
+
],
|
| 18 |
+
"max_position_embeddings": 32768,
|
| 19 |
+
"max_window_layers": 1,
|
| 20 |
+
"model_type": "qwen2",
|
| 21 |
+
"neuron": {
|
| 22 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 23 |
+
"batch_size": 1,
|
| 24 |
+
"capacity_factor": null,
|
| 25 |
+
"checkpoint_id": "yujiepan/qwen2.5-128k-tiny-random",
|
| 26 |
+
"checkpoint_revision": "c8296d4ca3f87782876d2382fbb6481d1beb8ef0",
|
| 27 |
+
"continuous_batching": false,
|
| 28 |
+
"ep_degree": 1,
|
| 29 |
+
"fused_qkv": false,
|
| 30 |
+
"glu_mlp": true,
|
| 31 |
+
"local_ranks_size": 2,
|
| 32 |
+
"max_batch_size": 1,
|
| 33 |
+
"max_context_length": 1024,
|
| 34 |
+
"max_topk": 256,
|
| 35 |
+
"n_active_tokens": 1024,
|
| 36 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 37 |
+
"on_device_sampling": true,
|
| 38 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 39 |
+
"output_logits": false,
|
| 40 |
+
"pp_degree": 1,
|
| 41 |
+
"sequence_length": 1024,
|
| 42 |
+
"speculation_length": 0,
|
| 43 |
+
"start_rank_id": 0,
|
| 44 |
+
"target": "trn1",
|
| 45 |
+
"torch_dtype": "bfloat16",
|
| 46 |
+
"tp_degree": 2
|
| 47 |
+
},
|
| 48 |
+
"num_attention_heads": 4,
|
| 49 |
+
"num_hidden_layers": 2,
|
| 50 |
+
"num_key_value_heads": 2,
|
| 51 |
+
"rms_norm_eps": 1e-06,
|
| 52 |
+
"rope_scaling": {
|
| 53 |
+
"factor": 4.0,
|
| 54 |
+
"original_max_position_embeddings": 32768,
|
| 55 |
+
"rope_type": "yarn",
|
| 56 |
+
"type": "yarn"
|
| 57 |
+
},
|
| 58 |
+
"rope_theta": 1000000.0,
|
| 59 |
+
"sliding_window": null,
|
| 60 |
+
"tie_word_embeddings": false,
|
| 61 |
+
"use_cache": true,
|
| 62 |
+
"use_sliding_window": false,
|
| 63 |
+
"vocab_size": 152064
|
| 64 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/7f05bde17c7b0ffeb657897697f23d182f406b76ced7f1b2cd5741dc93fe2e2e/dadf01a9f544218eabdd.json
ADDED
|
@@ -0,0 +1,125 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "tiny-random/llama-4",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Llama4ForConditionalGeneration"
|
| 7 |
+
],
|
| 8 |
+
"boi_token_index": 200080,
|
| 9 |
+
"dtype": "bfloat16",
|
| 10 |
+
"eoi_token_index": 200081,
|
| 11 |
+
"image_token_index": 200092,
|
| 12 |
+
"model_type": "llama4",
|
| 13 |
+
"neuron": {
|
| 14 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 15 |
+
"batch_size": 1,
|
| 16 |
+
"capacity_factor": null,
|
| 17 |
+
"checkpoint_id": "tiny-random/llama-4",
|
| 18 |
+
"checkpoint_revision": "9e716f5d4d1ffe0a44a15f46f4a12b840439aba4",
|
| 19 |
+
"continuous_batching": false,
|
| 20 |
+
"ep_degree": 1,
|
| 21 |
+
"fused_qkv": false,
|
| 22 |
+
"glu_mlp": true,
|
| 23 |
+
"local_ranks_size": 2,
|
| 24 |
+
"max_batch_size": 1,
|
| 25 |
+
"max_context_length": 1024,
|
| 26 |
+
"max_topk": 256,
|
| 27 |
+
"n_active_tokens": 1024,
|
| 28 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 29 |
+
"on_device_sampling": true,
|
| 30 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 31 |
+
"output_logits": false,
|
| 32 |
+
"pp_degree": 1,
|
| 33 |
+
"sequence_length": 1024,
|
| 34 |
+
"speculation_length": 0,
|
| 35 |
+
"start_rank_id": 0,
|
| 36 |
+
"target": "trn1",
|
| 37 |
+
"torch_dtype": "bfloat16",
|
| 38 |
+
"tp_degree": 2
|
| 39 |
+
},
|
| 40 |
+
"text_config": {
|
| 41 |
+
"_attn_implementation_autoset": true,
|
| 42 |
+
"attention_bias": false,
|
| 43 |
+
"attention_chunk_size": 128,
|
| 44 |
+
"attention_dropout": 0.0,
|
| 45 |
+
"attn_scale": 0.1,
|
| 46 |
+
"attn_temperature_tuning": 4,
|
| 47 |
+
"bos_token_id": 200000,
|
| 48 |
+
"cache_implementation": "hybrid",
|
| 49 |
+
"dtype": "bfloat16",
|
| 50 |
+
"eos_token_id": [
|
| 51 |
+
200001,
|
| 52 |
+
200007,
|
| 53 |
+
200008
|
| 54 |
+
],
|
| 55 |
+
"floor_scale": 8192,
|
| 56 |
+
"for_llm_compressor": false,
|
| 57 |
+
"head_dim": 32,
|
| 58 |
+
"hidden_act": "silu",
|
| 59 |
+
"hidden_size": 32,
|
| 60 |
+
"initializer_range": 0.02,
|
| 61 |
+
"interleave_moe_layer_step": 2,
|
| 62 |
+
"intermediate_size": 64,
|
| 63 |
+
"intermediate_size_mlp": 128,
|
| 64 |
+
"layer_types": [
|
| 65 |
+
"chunked_attention",
|
| 66 |
+
"chunked_attention",
|
| 67 |
+
"chunked_attention",
|
| 68 |
+
"full_attention"
|
| 69 |
+
],
|
| 70 |
+
"max_position_embeddings": 1048576,
|
| 71 |
+
"model_type": "llama4_text",
|
| 72 |
+
"moe_layers": [
|
| 73 |
+
1,
|
| 74 |
+
3
|
| 75 |
+
],
|
| 76 |
+
"no_rope_layers": [
|
| 77 |
+
1,
|
| 78 |
+
1,
|
| 79 |
+
1,
|
| 80 |
+
0
|
| 81 |
+
],
|
| 82 |
+
"num_attention_heads": 1,
|
| 83 |
+
"num_experts_per_tok": 1,
|
| 84 |
+
"num_hidden_layers": 4,
|
| 85 |
+
"num_key_value_heads": 1,
|
| 86 |
+
"num_local_experts": 8,
|
| 87 |
+
"output_router_logits": false,
|
| 88 |
+
"pad_token_id": 200018,
|
| 89 |
+
"rms_norm_eps": 1e-05,
|
| 90 |
+
"rope_scaling": null,
|
| 91 |
+
"rope_theta": 500000.0,
|
| 92 |
+
"router_aux_loss_coef": 0.001,
|
| 93 |
+
"router_jitter_noise": 0.0,
|
| 94 |
+
"tie_word_embeddings": true,
|
| 95 |
+
"use_cache": true,
|
| 96 |
+
"use_qk_norm": true,
|
| 97 |
+
"vocab_size": 202048
|
| 98 |
+
},
|
| 99 |
+
"tie_word_embeddings": false,
|
| 100 |
+
"vision_config": {
|
| 101 |
+
"_attn_implementation_autoset": true,
|
| 102 |
+
"_vision_feature_layer": -1,
|
| 103 |
+
"attention_dropout": 0.0,
|
| 104 |
+
"hidden_act": "gelu",
|
| 105 |
+
"hidden_size": 32,
|
| 106 |
+
"image_size": 336,
|
| 107 |
+
"initializer_range": 0.02,
|
| 108 |
+
"intermediate_size": 128,
|
| 109 |
+
"model_type": "llama4_vision_model",
|
| 110 |
+
"multi_modal_projector_bias": false,
|
| 111 |
+
"norm_eps": 1e-05,
|
| 112 |
+
"num_attention_heads": 1,
|
| 113 |
+
"num_channels": 3,
|
| 114 |
+
"num_hidden_layers": 2,
|
| 115 |
+
"patch_size": 14,
|
| 116 |
+
"pixel_shuffle_ratio": 0.5,
|
| 117 |
+
"projector_dropout": 0.0,
|
| 118 |
+
"projector_input_dim": 32,
|
| 119 |
+
"projector_output_dim": 32,
|
| 120 |
+
"rope_theta": 10000,
|
| 121 |
+
"vision_feature_layer": -1,
|
| 122 |
+
"vision_feature_select_strategy": "default",
|
| 123 |
+
"vision_output_dim": 32
|
| 124 |
+
}
|
| 125 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/8c90ac2593ed0b7f1ecb60e82cb184fb11f2ea640befa1cc7b10766a5c02525d/3c497b71919d297a9da9.json
ADDED
|
@@ -0,0 +1,164 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "microsoft/Phi-3.5-mini-instruct",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Phi3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"auto_map": {
|
| 11 |
+
"AutoConfig": "configuration_phi3.Phi3Config",
|
| 12 |
+
"AutoModelForCausalLM": "modeling_phi3.Phi3ForCausalLM"
|
| 13 |
+
},
|
| 14 |
+
"dtype": "bfloat16",
|
| 15 |
+
"embd_pdrop": 0.0,
|
| 16 |
+
"hidden_act": "silu",
|
| 17 |
+
"hidden_size": 3072,
|
| 18 |
+
"initializer_range": 0.02,
|
| 19 |
+
"intermediate_size": 8192,
|
| 20 |
+
"max_position_embeddings": 131072,
|
| 21 |
+
"model_type": "phi3",
|
| 22 |
+
"neuron": {
|
| 23 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 24 |
+
"batch_size": 1,
|
| 25 |
+
"capacity_factor": null,
|
| 26 |
+
"checkpoint_id": "microsoft/Phi-3.5-mini-instruct",
|
| 27 |
+
"checkpoint_revision": "2fe192450127e6a83f7441aef6e3ca586c338b77",
|
| 28 |
+
"continuous_batching": false,
|
| 29 |
+
"ep_degree": 1,
|
| 30 |
+
"fused_qkv": true,
|
| 31 |
+
"glu_mlp": true,
|
| 32 |
+
"local_ranks_size": 2,
|
| 33 |
+
"max_batch_size": 1,
|
| 34 |
+
"max_context_length": 8192,
|
| 35 |
+
"max_topk": 256,
|
| 36 |
+
"n_active_tokens": 8192,
|
| 37 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 38 |
+
"on_device_sampling": true,
|
| 39 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 40 |
+
"output_logits": false,
|
| 41 |
+
"pp_degree": 1,
|
| 42 |
+
"sequence_length": 8192,
|
| 43 |
+
"speculation_length": 0,
|
| 44 |
+
"start_rank_id": 0,
|
| 45 |
+
"target": "trn1",
|
| 46 |
+
"torch_dtype": "bfloat16",
|
| 47 |
+
"tp_degree": 2
|
| 48 |
+
},
|
| 49 |
+
"num_attention_heads": 32,
|
| 50 |
+
"num_hidden_layers": 32,
|
| 51 |
+
"num_key_value_heads": 32,
|
| 52 |
+
"original_max_position_embeddings": 4096,
|
| 53 |
+
"partial_rotary_factor": 1.0,
|
| 54 |
+
"resid_pdrop": 0.0,
|
| 55 |
+
"rms_norm_eps": 1e-05,
|
| 56 |
+
"rope_scaling": {
|
| 57 |
+
"long_factor": [
|
| 58 |
+
1.0800000429153442,
|
| 59 |
+
1.1100000143051147,
|
| 60 |
+
1.1399999856948853,
|
| 61 |
+
1.340000033378601,
|
| 62 |
+
1.5899999141693115,
|
| 63 |
+
1.600000023841858,
|
| 64 |
+
1.6200000047683716,
|
| 65 |
+
2.620000123977661,
|
| 66 |
+
3.2300000190734863,
|
| 67 |
+
3.2300000190734863,
|
| 68 |
+
4.789999961853027,
|
| 69 |
+
7.400000095367432,
|
| 70 |
+
7.700000286102295,
|
| 71 |
+
9.09000015258789,
|
| 72 |
+
12.199999809265137,
|
| 73 |
+
17.670000076293945,
|
| 74 |
+
24.46000099182129,
|
| 75 |
+
28.57000160217285,
|
| 76 |
+
30.420001983642578,
|
| 77 |
+
30.840002059936523,
|
| 78 |
+
32.590003967285156,
|
| 79 |
+
32.93000411987305,
|
| 80 |
+
42.320003509521484,
|
| 81 |
+
44.96000289916992,
|
| 82 |
+
50.340003967285156,
|
| 83 |
+
50.45000457763672,
|
| 84 |
+
57.55000305175781,
|
| 85 |
+
57.93000411987305,
|
| 86 |
+
58.21000289916992,
|
| 87 |
+
60.1400032043457,
|
| 88 |
+
62.61000442504883,
|
| 89 |
+
62.62000274658203,
|
| 90 |
+
62.71000289916992,
|
| 91 |
+
63.1400032043457,
|
| 92 |
+
63.1400032043457,
|
| 93 |
+
63.77000427246094,
|
| 94 |
+
63.93000411987305,
|
| 95 |
+
63.96000289916992,
|
| 96 |
+
63.970001220703125,
|
| 97 |
+
64.02999877929688,
|
| 98 |
+
64.06999969482422,
|
| 99 |
+
64.08000183105469,
|
| 100 |
+
64.12000274658203,
|
| 101 |
+
64.41000366210938,
|
| 102 |
+
64.4800033569336,
|
| 103 |
+
64.51000213623047,
|
| 104 |
+
64.52999877929688,
|
| 105 |
+
64.83999633789062
|
| 106 |
+
],
|
| 107 |
+
"short_factor": [
|
| 108 |
+
1.0,
|
| 109 |
+
1.0199999809265137,
|
| 110 |
+
1.0299999713897705,
|
| 111 |
+
1.0299999713897705,
|
| 112 |
+
1.0499999523162842,
|
| 113 |
+
1.0499999523162842,
|
| 114 |
+
1.0499999523162842,
|
| 115 |
+
1.0499999523162842,
|
| 116 |
+
1.0499999523162842,
|
| 117 |
+
1.0699999332427979,
|
| 118 |
+
1.0999999046325684,
|
| 119 |
+
1.1099998950958252,
|
| 120 |
+
1.1599998474121094,
|
| 121 |
+
1.1599998474121094,
|
| 122 |
+
1.1699998378753662,
|
| 123 |
+
1.2899998426437378,
|
| 124 |
+
1.339999794960022,
|
| 125 |
+
1.679999828338623,
|
| 126 |
+
1.7899998426437378,
|
| 127 |
+
1.8199998140335083,
|
| 128 |
+
1.8499997854232788,
|
| 129 |
+
1.8799997568130493,
|
| 130 |
+
1.9099997282028198,
|
| 131 |
+
1.9399996995925903,
|
| 132 |
+
1.9899996519088745,
|
| 133 |
+
2.0199997425079346,
|
| 134 |
+
2.0199997425079346,
|
| 135 |
+
2.0199997425079346,
|
| 136 |
+
2.0199997425079346,
|
| 137 |
+
2.0199997425079346,
|
| 138 |
+
2.0199997425079346,
|
| 139 |
+
2.0299997329711914,
|
| 140 |
+
2.0299997329711914,
|
| 141 |
+
2.0299997329711914,
|
| 142 |
+
2.0299997329711914,
|
| 143 |
+
2.0299997329711914,
|
| 144 |
+
2.0299997329711914,
|
| 145 |
+
2.0299997329711914,
|
| 146 |
+
2.0299997329711914,
|
| 147 |
+
2.0299997329711914,
|
| 148 |
+
2.0799996852874756,
|
| 149 |
+
2.0899996757507324,
|
| 150 |
+
2.189999580383301,
|
| 151 |
+
2.2199995517730713,
|
| 152 |
+
2.5899994373321533,
|
| 153 |
+
2.729999542236328,
|
| 154 |
+
2.749999523162842,
|
| 155 |
+
2.8399994373321533
|
| 156 |
+
],
|
| 157 |
+
"type": "longrope"
|
| 158 |
+
},
|
| 159 |
+
"rope_theta": 10000.0,
|
| 160 |
+
"sliding_window": 262144,
|
| 161 |
+
"tie_word_embeddings": false,
|
| 162 |
+
"use_cache": true,
|
| 163 |
+
"vocab_size": 32064
|
| 164 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/8c90ac2593ed0b7f1ecb60e82cb184fb11f2ea640befa1cc7b10766a5c02525d/d00ca5f7300e7c2696dd.json
ADDED
|
@@ -0,0 +1,164 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "microsoft/Phi-3.5-mini-instruct",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Phi3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"auto_map": {
|
| 11 |
+
"AutoConfig": "configuration_phi3.Phi3Config",
|
| 12 |
+
"AutoModelForCausalLM": "modeling_phi3.Phi3ForCausalLM"
|
| 13 |
+
},
|
| 14 |
+
"dtype": "bfloat16",
|
| 15 |
+
"embd_pdrop": 0.0,
|
| 16 |
+
"hidden_act": "silu",
|
| 17 |
+
"hidden_size": 3072,
|
| 18 |
+
"initializer_range": 0.02,
|
| 19 |
+
"intermediate_size": 8192,
|
| 20 |
+
"max_position_embeddings": 131072,
|
| 21 |
+
"model_type": "phi3",
|
| 22 |
+
"neuron": {
|
| 23 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 24 |
+
"batch_size": 4,
|
| 25 |
+
"capacity_factor": null,
|
| 26 |
+
"checkpoint_id": "microsoft/Phi-3.5-mini-instruct",
|
| 27 |
+
"checkpoint_revision": "2fe192450127e6a83f7441aef6e3ca586c338b77",
|
| 28 |
+
"continuous_batching": true,
|
| 29 |
+
"ep_degree": 1,
|
| 30 |
+
"fused_qkv": true,
|
| 31 |
+
"glu_mlp": true,
|
| 32 |
+
"local_ranks_size": 2,
|
| 33 |
+
"max_batch_size": 4,
|
| 34 |
+
"max_context_length": 4096,
|
| 35 |
+
"max_topk": 256,
|
| 36 |
+
"n_active_tokens": 4096,
|
| 37 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 38 |
+
"on_device_sampling": true,
|
| 39 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 40 |
+
"output_logits": false,
|
| 41 |
+
"pp_degree": 1,
|
| 42 |
+
"sequence_length": 4096,
|
| 43 |
+
"speculation_length": 0,
|
| 44 |
+
"start_rank_id": 0,
|
| 45 |
+
"target": "trn1",
|
| 46 |
+
"torch_dtype": "bfloat16",
|
| 47 |
+
"tp_degree": 2
|
| 48 |
+
},
|
| 49 |
+
"num_attention_heads": 32,
|
| 50 |
+
"num_hidden_layers": 32,
|
| 51 |
+
"num_key_value_heads": 32,
|
| 52 |
+
"original_max_position_embeddings": 4096,
|
| 53 |
+
"partial_rotary_factor": 1.0,
|
| 54 |
+
"resid_pdrop": 0.0,
|
| 55 |
+
"rms_norm_eps": 1e-05,
|
| 56 |
+
"rope_scaling": {
|
| 57 |
+
"long_factor": [
|
| 58 |
+
1.0800000429153442,
|
| 59 |
+
1.1100000143051147,
|
| 60 |
+
1.1399999856948853,
|
| 61 |
+
1.340000033378601,
|
| 62 |
+
1.5899999141693115,
|
| 63 |
+
1.600000023841858,
|
| 64 |
+
1.6200000047683716,
|
| 65 |
+
2.620000123977661,
|
| 66 |
+
3.2300000190734863,
|
| 67 |
+
3.2300000190734863,
|
| 68 |
+
4.789999961853027,
|
| 69 |
+
7.400000095367432,
|
| 70 |
+
7.700000286102295,
|
| 71 |
+
9.09000015258789,
|
| 72 |
+
12.199999809265137,
|
| 73 |
+
17.670000076293945,
|
| 74 |
+
24.46000099182129,
|
| 75 |
+
28.57000160217285,
|
| 76 |
+
30.420001983642578,
|
| 77 |
+
30.840002059936523,
|
| 78 |
+
32.590003967285156,
|
| 79 |
+
32.93000411987305,
|
| 80 |
+
42.320003509521484,
|
| 81 |
+
44.96000289916992,
|
| 82 |
+
50.340003967285156,
|
| 83 |
+
50.45000457763672,
|
| 84 |
+
57.55000305175781,
|
| 85 |
+
57.93000411987305,
|
| 86 |
+
58.21000289916992,
|
| 87 |
+
60.1400032043457,
|
| 88 |
+
62.61000442504883,
|
| 89 |
+
62.62000274658203,
|
| 90 |
+
62.71000289916992,
|
| 91 |
+
63.1400032043457,
|
| 92 |
+
63.1400032043457,
|
| 93 |
+
63.77000427246094,
|
| 94 |
+
63.93000411987305,
|
| 95 |
+
63.96000289916992,
|
| 96 |
+
63.970001220703125,
|
| 97 |
+
64.02999877929688,
|
| 98 |
+
64.06999969482422,
|
| 99 |
+
64.08000183105469,
|
| 100 |
+
64.12000274658203,
|
| 101 |
+
64.41000366210938,
|
| 102 |
+
64.4800033569336,
|
| 103 |
+
64.51000213623047,
|
| 104 |
+
64.52999877929688,
|
| 105 |
+
64.83999633789062
|
| 106 |
+
],
|
| 107 |
+
"short_factor": [
|
| 108 |
+
1.0,
|
| 109 |
+
1.0199999809265137,
|
| 110 |
+
1.0299999713897705,
|
| 111 |
+
1.0299999713897705,
|
| 112 |
+
1.0499999523162842,
|
| 113 |
+
1.0499999523162842,
|
| 114 |
+
1.0499999523162842,
|
| 115 |
+
1.0499999523162842,
|
| 116 |
+
1.0499999523162842,
|
| 117 |
+
1.0699999332427979,
|
| 118 |
+
1.0999999046325684,
|
| 119 |
+
1.1099998950958252,
|
| 120 |
+
1.1599998474121094,
|
| 121 |
+
1.1599998474121094,
|
| 122 |
+
1.1699998378753662,
|
| 123 |
+
1.2899998426437378,
|
| 124 |
+
1.339999794960022,
|
| 125 |
+
1.679999828338623,
|
| 126 |
+
1.7899998426437378,
|
| 127 |
+
1.8199998140335083,
|
| 128 |
+
1.8499997854232788,
|
| 129 |
+
1.8799997568130493,
|
| 130 |
+
1.9099997282028198,
|
| 131 |
+
1.9399996995925903,
|
| 132 |
+
1.9899996519088745,
|
| 133 |
+
2.0199997425079346,
|
| 134 |
+
2.0199997425079346,
|
| 135 |
+
2.0199997425079346,
|
| 136 |
+
2.0199997425079346,
|
| 137 |
+
2.0199997425079346,
|
| 138 |
+
2.0199997425079346,
|
| 139 |
+
2.0299997329711914,
|
| 140 |
+
2.0299997329711914,
|
| 141 |
+
2.0299997329711914,
|
| 142 |
+
2.0299997329711914,
|
| 143 |
+
2.0299997329711914,
|
| 144 |
+
2.0299997329711914,
|
| 145 |
+
2.0299997329711914,
|
| 146 |
+
2.0299997329711914,
|
| 147 |
+
2.0299997329711914,
|
| 148 |
+
2.0799996852874756,
|
| 149 |
+
2.0899996757507324,
|
| 150 |
+
2.189999580383301,
|
| 151 |
+
2.2199995517730713,
|
| 152 |
+
2.5899994373321533,
|
| 153 |
+
2.729999542236328,
|
| 154 |
+
2.749999523162842,
|
| 155 |
+
2.8399994373321533
|
| 156 |
+
],
|
| 157 |
+
"type": "longrope"
|
| 158 |
+
},
|
| 159 |
+
"rope_theta": 10000.0,
|
| 160 |
+
"sliding_window": 262144,
|
| 161 |
+
"tie_word_embeddings": false,
|
| 162 |
+
"use_cache": true,
|
| 163 |
+
"vocab_size": 32064
|
| 164 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/920f44ce6d3e004d1ce547ae06644f7be262180644b04573153aa15d98742edc/b5678f2b1f926f36a4fd.json
ADDED
|
@@ -0,0 +1,65 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "optimum-internal-testing/tiny-random-qwen3_moe",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3MoeForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"decoder_sparse_step": 2,
|
| 11 |
+
"dtype": "float32",
|
| 12 |
+
"head_dim": 32,
|
| 13 |
+
"hidden_act": "silu",
|
| 14 |
+
"hidden_size": 64,
|
| 15 |
+
"initializer_range": 0.02,
|
| 16 |
+
"intermediate_size": 128,
|
| 17 |
+
"max_position_embeddings": 40960,
|
| 18 |
+
"max_window_layers": 1,
|
| 19 |
+
"mlp_only_layers": [],
|
| 20 |
+
"model_type": "qwen3_moe",
|
| 21 |
+
"moe_intermediate_size": 128,
|
| 22 |
+
"neuron": {
|
| 23 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 24 |
+
"batch_size": 1,
|
| 25 |
+
"capacity_factor": null,
|
| 26 |
+
"checkpoint_id": "optimum-internal-testing/tiny-random-qwen3_moe",
|
| 27 |
+
"checkpoint_revision": "e0230be2839556b44b7400a233c73c74b4abb7af",
|
| 28 |
+
"continuous_batching": false,
|
| 29 |
+
"ep_degree": 1,
|
| 30 |
+
"fused_qkv": false,
|
| 31 |
+
"glu_mlp": true,
|
| 32 |
+
"local_ranks_size": 2,
|
| 33 |
+
"max_batch_size": 1,
|
| 34 |
+
"max_context_length": 1024,
|
| 35 |
+
"max_topk": 256,
|
| 36 |
+
"n_active_tokens": 1024,
|
| 37 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 38 |
+
"on_device_sampling": true,
|
| 39 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 40 |
+
"output_logits": false,
|
| 41 |
+
"pp_degree": 1,
|
| 42 |
+
"sequence_length": 1024,
|
| 43 |
+
"speculation_length": 0,
|
| 44 |
+
"start_rank_id": 0,
|
| 45 |
+
"target": "trn1",
|
| 46 |
+
"torch_dtype": "float32",
|
| 47 |
+
"tp_degree": 2
|
| 48 |
+
},
|
| 49 |
+
"norm_topk_prob": true,
|
| 50 |
+
"num_attention_heads": 2,
|
| 51 |
+
"num_experts": 8,
|
| 52 |
+
"num_experts_per_tok": 2,
|
| 53 |
+
"num_hidden_layers": 2,
|
| 54 |
+
"num_key_value_heads": 1,
|
| 55 |
+
"output_router_logits": false,
|
| 56 |
+
"rms_norm_eps": 1e-06,
|
| 57 |
+
"rope_scaling": null,
|
| 58 |
+
"rope_theta": 1000000.0,
|
| 59 |
+
"router_aux_loss_coef": 0.001,
|
| 60 |
+
"sliding_window": null,
|
| 61 |
+
"tie_word_embeddings": true,
|
| 62 |
+
"use_cache": true,
|
| 63 |
+
"use_sliding_window": false,
|
| 64 |
+
"vocab_size": 151936
|
| 65 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/929b02754a13cbfdf657d863c3fc6f3bce672879bc6ae48ab45be21e881e9ec2/21441e8ed07d8b61298d.json
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-0.6B",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 1024,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 3072,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention"
|
| 45 |
+
],
|
| 46 |
+
"max_position_embeddings": 40960,
|
| 47 |
+
"max_window_layers": 28,
|
| 48 |
+
"model_type": "qwen3",
|
| 49 |
+
"neuron": {
|
| 50 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 51 |
+
"batch_size": 1,
|
| 52 |
+
"capacity_factor": null,
|
| 53 |
+
"checkpoint_id": "Qwen/Qwen3-0.6B",
|
| 54 |
+
"checkpoint_revision": "c1899de289a04d12100db370d81485cdf75e47ca",
|
| 55 |
+
"continuous_batching": false,
|
| 56 |
+
"ep_degree": 1,
|
| 57 |
+
"fused_qkv": true,
|
| 58 |
+
"glu_mlp": true,
|
| 59 |
+
"local_ranks_size": 2,
|
| 60 |
+
"max_batch_size": 1,
|
| 61 |
+
"max_context_length": 8192,
|
| 62 |
+
"max_topk": 256,
|
| 63 |
+
"n_active_tokens": 8192,
|
| 64 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 65 |
+
"on_device_sampling": true,
|
| 66 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 67 |
+
"output_logits": false,
|
| 68 |
+
"pp_degree": 1,
|
| 69 |
+
"sequence_length": 8192,
|
| 70 |
+
"speculation_length": 0,
|
| 71 |
+
"start_rank_id": 0,
|
| 72 |
+
"target": "trn1",
|
| 73 |
+
"torch_dtype": "bfloat16",
|
| 74 |
+
"tp_degree": 2
|
| 75 |
+
},
|
| 76 |
+
"num_attention_heads": 16,
|
| 77 |
+
"num_hidden_layers": 28,
|
| 78 |
+
"num_key_value_heads": 8,
|
| 79 |
+
"rms_norm_eps": 1e-06,
|
| 80 |
+
"rope_scaling": null,
|
| 81 |
+
"rope_theta": 1000000,
|
| 82 |
+
"sliding_window": null,
|
| 83 |
+
"tie_word_embeddings": true,
|
| 84 |
+
"use_cache": true,
|
| 85 |
+
"use_sliding_window": false,
|
| 86 |
+
"vocab_size": 151936
|
| 87 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/929b02754a13cbfdf657d863c3fc6f3bce672879bc6ae48ab45be21e881e9ec2/f65f144a780153ddd757.json
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-0.6B",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 1024,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 3072,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention"
|
| 45 |
+
],
|
| 46 |
+
"max_position_embeddings": 40960,
|
| 47 |
+
"max_window_layers": 28,
|
| 48 |
+
"model_type": "qwen3",
|
| 49 |
+
"neuron": {
|
| 50 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 51 |
+
"batch_size": 4,
|
| 52 |
+
"capacity_factor": null,
|
| 53 |
+
"checkpoint_id": "Qwen/Qwen3-0.6B",
|
| 54 |
+
"checkpoint_revision": "c1899de289a04d12100db370d81485cdf75e47ca",
|
| 55 |
+
"continuous_batching": true,
|
| 56 |
+
"ep_degree": 1,
|
| 57 |
+
"fused_qkv": true,
|
| 58 |
+
"glu_mlp": true,
|
| 59 |
+
"local_ranks_size": 2,
|
| 60 |
+
"max_batch_size": 4,
|
| 61 |
+
"max_context_length": 4096,
|
| 62 |
+
"max_topk": 256,
|
| 63 |
+
"n_active_tokens": 4096,
|
| 64 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 65 |
+
"on_device_sampling": false,
|
| 66 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 67 |
+
"output_logits": false,
|
| 68 |
+
"pp_degree": 1,
|
| 69 |
+
"sequence_length": 4096,
|
| 70 |
+
"speculation_length": 0,
|
| 71 |
+
"start_rank_id": 0,
|
| 72 |
+
"target": "trn1",
|
| 73 |
+
"torch_dtype": "bfloat16",
|
| 74 |
+
"tp_degree": 2
|
| 75 |
+
},
|
| 76 |
+
"num_attention_heads": 16,
|
| 77 |
+
"num_hidden_layers": 28,
|
| 78 |
+
"num_key_value_heads": 8,
|
| 79 |
+
"rms_norm_eps": 1e-06,
|
| 80 |
+
"rope_scaling": null,
|
| 81 |
+
"rope_theta": 1000000,
|
| 82 |
+
"sliding_window": null,
|
| 83 |
+
"tie_word_embeddings": true,
|
| 84 |
+
"use_cache": true,
|
| 85 |
+
"use_sliding_window": false,
|
| 86 |
+
"vocab_size": 151936
|
| 87 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/4d99c6a74830655285f6.json
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "unsloth/Llama-3.2-1B-Instruct",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"LlamaForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 64,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 2048,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 8192,
|
| 16 |
+
"max_position_embeddings": 131072,
|
| 17 |
+
"mlp_bias": false,
|
| 18 |
+
"model_type": "llama",
|
| 19 |
+
"neuron": {
|
| 20 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 21 |
+
"batch_size": 4,
|
| 22 |
+
"capacity_factor": null,
|
| 23 |
+
"checkpoint_id": "unsloth/Llama-3.2-1B-Instruct",
|
| 24 |
+
"checkpoint_revision": "5a8abab4a5d6f164389b1079fb721cfab8d7126c",
|
| 25 |
+
"continuous_batching": true,
|
| 26 |
+
"ep_degree": 1,
|
| 27 |
+
"fused_qkv": true,
|
| 28 |
+
"glu_mlp": true,
|
| 29 |
+
"local_ranks_size": 2,
|
| 30 |
+
"max_batch_size": 4,
|
| 31 |
+
"max_context_length": 4096,
|
| 32 |
+
"max_topk": 256,
|
| 33 |
+
"n_active_tokens": 4096,
|
| 34 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 35 |
+
"on_device_sampling": true,
|
| 36 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 37 |
+
"output_logits": false,
|
| 38 |
+
"pp_degree": 1,
|
| 39 |
+
"sequence_length": 4096,
|
| 40 |
+
"speculation_length": 0,
|
| 41 |
+
"start_rank_id": 0,
|
| 42 |
+
"target": "trn1",
|
| 43 |
+
"torch_dtype": "bfloat16",
|
| 44 |
+
"tp_degree": 2
|
| 45 |
+
},
|
| 46 |
+
"num_attention_heads": 32,
|
| 47 |
+
"num_hidden_layers": 16,
|
| 48 |
+
"num_key_value_heads": 8,
|
| 49 |
+
"pretraining_tp": 1,
|
| 50 |
+
"rms_norm_eps": 1e-05,
|
| 51 |
+
"rope_scaling": {
|
| 52 |
+
"factor": 32.0,
|
| 53 |
+
"high_freq_factor": 4.0,
|
| 54 |
+
"low_freq_factor": 1.0,
|
| 55 |
+
"original_max_position_embeddings": 8192,
|
| 56 |
+
"rope_type": "llama3"
|
| 57 |
+
},
|
| 58 |
+
"rope_theta": 500000.0,
|
| 59 |
+
"tie_word_embeddings": true,
|
| 60 |
+
"unsloth_fixed": true,
|
| 61 |
+
"use_cache": true,
|
| 62 |
+
"vocab_size": 128256
|
| 63 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/cb2c5b9dc81e576d83aa.json
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "unsloth/Llama-3.2-1B-Instruct",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"LlamaForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 64,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 2048,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 8192,
|
| 16 |
+
"max_position_embeddings": 131072,
|
| 17 |
+
"mlp_bias": false,
|
| 18 |
+
"model_type": "llama",
|
| 19 |
+
"neuron": {
|
| 20 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 21 |
+
"batch_size": 1,
|
| 22 |
+
"capacity_factor": null,
|
| 23 |
+
"checkpoint_id": "unsloth/Llama-3.2-1B-Instruct",
|
| 24 |
+
"checkpoint_revision": null,
|
| 25 |
+
"continuous_batching": false,
|
| 26 |
+
"ep_degree": 1,
|
| 27 |
+
"fused_qkv": false,
|
| 28 |
+
"glu_mlp": true,
|
| 29 |
+
"local_ranks_size": 2,
|
| 30 |
+
"max_batch_size": 1,
|
| 31 |
+
"max_context_length": 4096,
|
| 32 |
+
"max_topk": 256,
|
| 33 |
+
"n_active_tokens": 4096,
|
| 34 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 35 |
+
"on_device_sampling": false,
|
| 36 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 37 |
+
"output_logits": false,
|
| 38 |
+
"pp_degree": 1,
|
| 39 |
+
"sequence_length": 4096,
|
| 40 |
+
"speculation_length": 5,
|
| 41 |
+
"start_rank_id": 0,
|
| 42 |
+
"target": "trn1",
|
| 43 |
+
"torch_dtype": "bfloat16",
|
| 44 |
+
"tp_degree": 2
|
| 45 |
+
},
|
| 46 |
+
"num_attention_heads": 32,
|
| 47 |
+
"num_hidden_layers": 16,
|
| 48 |
+
"num_key_value_heads": 8,
|
| 49 |
+
"pretraining_tp": 1,
|
| 50 |
+
"rms_norm_eps": 1e-05,
|
| 51 |
+
"rope_scaling": {
|
| 52 |
+
"factor": 32.0,
|
| 53 |
+
"high_freq_factor": 4.0,
|
| 54 |
+
"low_freq_factor": 1.0,
|
| 55 |
+
"original_max_position_embeddings": 8192,
|
| 56 |
+
"rope_type": "llama3"
|
| 57 |
+
},
|
| 58 |
+
"rope_theta": 500000.0,
|
| 59 |
+
"tie_word_embeddings": true,
|
| 60 |
+
"unsloth_fixed": true,
|
| 61 |
+
"use_cache": true,
|
| 62 |
+
"vocab_size": 128256
|
| 63 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/d020f018e0819410feb2.json
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "unsloth/Llama-3.2-1B-Instruct",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"LlamaForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 64,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 2048,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 8192,
|
| 16 |
+
"max_position_embeddings": 131072,
|
| 17 |
+
"mlp_bias": false,
|
| 18 |
+
"model_type": "llama",
|
| 19 |
+
"neuron": {
|
| 20 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 21 |
+
"batch_size": 1,
|
| 22 |
+
"capacity_factor": null,
|
| 23 |
+
"checkpoint_id": "unsloth/Llama-3.2-1B-Instruct",
|
| 24 |
+
"checkpoint_revision": null,
|
| 25 |
+
"continuous_batching": false,
|
| 26 |
+
"ep_degree": 1,
|
| 27 |
+
"fused_qkv": false,
|
| 28 |
+
"glu_mlp": true,
|
| 29 |
+
"local_ranks_size": 2,
|
| 30 |
+
"max_batch_size": 1,
|
| 31 |
+
"max_context_length": 4096,
|
| 32 |
+
"max_topk": 256,
|
| 33 |
+
"n_active_tokens": 4096,
|
| 34 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 35 |
+
"on_device_sampling": false,
|
| 36 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 37 |
+
"output_logits": false,
|
| 38 |
+
"pp_degree": 1,
|
| 39 |
+
"sequence_length": 4096,
|
| 40 |
+
"speculation_length": 0,
|
| 41 |
+
"start_rank_id": 0,
|
| 42 |
+
"target": "trn1",
|
| 43 |
+
"torch_dtype": "bfloat16",
|
| 44 |
+
"tp_degree": 2
|
| 45 |
+
},
|
| 46 |
+
"num_attention_heads": 32,
|
| 47 |
+
"num_hidden_layers": 16,
|
| 48 |
+
"num_key_value_heads": 8,
|
| 49 |
+
"pretraining_tp": 1,
|
| 50 |
+
"rms_norm_eps": 1e-05,
|
| 51 |
+
"rope_scaling": {
|
| 52 |
+
"factor": 32.0,
|
| 53 |
+
"high_freq_factor": 4.0,
|
| 54 |
+
"low_freq_factor": 1.0,
|
| 55 |
+
"original_max_position_embeddings": 8192,
|
| 56 |
+
"rope_type": "llama3"
|
| 57 |
+
},
|
| 58 |
+
"rope_theta": 500000.0,
|
| 59 |
+
"tie_word_embeddings": true,
|
| 60 |
+
"unsloth_fixed": true,
|
| 61 |
+
"use_cache": true,
|
| 62 |
+
"vocab_size": 128256
|
| 63 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/da4adf3105368a5df618.json
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "unsloth/Llama-3.2-1B-Instruct",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"LlamaForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 64,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 2048,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 8192,
|
| 16 |
+
"max_position_embeddings": 131072,
|
| 17 |
+
"mlp_bias": false,
|
| 18 |
+
"model_type": "llama",
|
| 19 |
+
"neuron": {
|
| 20 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 21 |
+
"batch_size": 1,
|
| 22 |
+
"capacity_factor": null,
|
| 23 |
+
"checkpoint_id": "unsloth/Llama-3.2-1B-Instruct",
|
| 24 |
+
"checkpoint_revision": "5a8abab4a5d6f164389b1079fb721cfab8d7126c",
|
| 25 |
+
"continuous_batching": false,
|
| 26 |
+
"ep_degree": 1,
|
| 27 |
+
"fused_qkv": true,
|
| 28 |
+
"glu_mlp": true,
|
| 29 |
+
"local_ranks_size": 2,
|
| 30 |
+
"max_batch_size": 1,
|
| 31 |
+
"max_context_length": 8192,
|
| 32 |
+
"max_topk": 256,
|
| 33 |
+
"n_active_tokens": 8192,
|
| 34 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 35 |
+
"on_device_sampling": true,
|
| 36 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 37 |
+
"output_logits": false,
|
| 38 |
+
"pp_degree": 1,
|
| 39 |
+
"sequence_length": 8192,
|
| 40 |
+
"speculation_length": 0,
|
| 41 |
+
"start_rank_id": 0,
|
| 42 |
+
"target": "trn1",
|
| 43 |
+
"torch_dtype": "bfloat16",
|
| 44 |
+
"tp_degree": 2
|
| 45 |
+
},
|
| 46 |
+
"num_attention_heads": 32,
|
| 47 |
+
"num_hidden_layers": 16,
|
| 48 |
+
"num_key_value_heads": 8,
|
| 49 |
+
"pretraining_tp": 1,
|
| 50 |
+
"rms_norm_eps": 1e-05,
|
| 51 |
+
"rope_scaling": {
|
| 52 |
+
"factor": 32.0,
|
| 53 |
+
"high_freq_factor": 4.0,
|
| 54 |
+
"low_freq_factor": 1.0,
|
| 55 |
+
"original_max_position_embeddings": 8192,
|
| 56 |
+
"rope_type": "llama3"
|
| 57 |
+
},
|
| 58 |
+
"rope_theta": 500000.0,
|
| 59 |
+
"tie_word_embeddings": true,
|
| 60 |
+
"unsloth_fixed": true,
|
| 61 |
+
"use_cache": true,
|
| 62 |
+
"vocab_size": 128256
|
| 63 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/d139acf64685f15794bb983ff6eb881bdd31304bae88b0ce1ed20a54c21f2265/0e7a6f2933f99785cba6.json
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "hf-internal-testing/tiny-random-GraniteForCausalLM",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"GraniteForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"attention_multiplier": 1.0,
|
| 11 |
+
"dtype": "float32",
|
| 12 |
+
"embedding_multiplier": 1.0,
|
| 13 |
+
"hidden_act": "silu",
|
| 14 |
+
"hidden_size": 32,
|
| 15 |
+
"initializer_range": 0.02,
|
| 16 |
+
"intermediate_size": 64,
|
| 17 |
+
"logits_scaling": 1.0,
|
| 18 |
+
"max_position_embeddings": 2048,
|
| 19 |
+
"mlp_bias": false,
|
| 20 |
+
"model_type": "granite",
|
| 21 |
+
"neuron": {
|
| 22 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 23 |
+
"batch_size": 1,
|
| 24 |
+
"capacity_factor": null,
|
| 25 |
+
"checkpoint_id": "hf-internal-testing/tiny-random-GraniteForCausalLM",
|
| 26 |
+
"checkpoint_revision": "c3074ebc0ac2fe545305f5e5f6cce2cc9b2aa0c5",
|
| 27 |
+
"continuous_batching": false,
|
| 28 |
+
"ep_degree": 1,
|
| 29 |
+
"fused_qkv": true,
|
| 30 |
+
"glu_mlp": true,
|
| 31 |
+
"local_ranks_size": 2,
|
| 32 |
+
"max_batch_size": 1,
|
| 33 |
+
"max_context_length": 1024,
|
| 34 |
+
"max_topk": 256,
|
| 35 |
+
"n_active_tokens": 1024,
|
| 36 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 37 |
+
"on_device_sampling": true,
|
| 38 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 39 |
+
"output_logits": false,
|
| 40 |
+
"pp_degree": 1,
|
| 41 |
+
"sequence_length": 1024,
|
| 42 |
+
"speculation_length": 0,
|
| 43 |
+
"start_rank_id": 0,
|
| 44 |
+
"target": "trn1",
|
| 45 |
+
"torch_dtype": "float32",
|
| 46 |
+
"tp_degree": 2
|
| 47 |
+
},
|
| 48 |
+
"num_attention_heads": 4,
|
| 49 |
+
"num_hidden_layers": 2,
|
| 50 |
+
"num_key_value_heads": 4,
|
| 51 |
+
"residual_multiplier": 1.0,
|
| 52 |
+
"rms_norm_eps": 1e-06,
|
| 53 |
+
"rope_scaling": null,
|
| 54 |
+
"rope_theta": 10000.0,
|
| 55 |
+
"tie_word_embeddings": false,
|
| 56 |
+
"use_cache": true,
|
| 57 |
+
"vocab_size": 49152
|
| 58 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/gemma3_text/unsloth/gemma-3-270m-it/51c77185b9832eaebdfc.json
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "unsloth/gemma-3-270m-it",
|
| 4 |
+
"_sliding_window_pattern": 6,
|
| 5 |
+
"_task": "text-generation",
|
| 6 |
+
"architectures": [
|
| 7 |
+
"Gemma3ForCausalLM"
|
| 8 |
+
],
|
| 9 |
+
"attention_bias": false,
|
| 10 |
+
"attention_dropout": 0.0,
|
| 11 |
+
"attn_logit_softcapping": null,
|
| 12 |
+
"dtype": "bfloat16",
|
| 13 |
+
"final_logit_softcapping": null,
|
| 14 |
+
"head_dim": 256,
|
| 15 |
+
"hidden_activation": "gelu_pytorch_tanh",
|
| 16 |
+
"hidden_size": 640,
|
| 17 |
+
"initializer_range": 0.02,
|
| 18 |
+
"intermediate_size": 2048,
|
| 19 |
+
"layer_types": [
|
| 20 |
+
"sliding_attention",
|
| 21 |
+
"sliding_attention",
|
| 22 |
+
"sliding_attention",
|
| 23 |
+
"sliding_attention",
|
| 24 |
+
"sliding_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"sliding_attention",
|
| 27 |
+
"sliding_attention",
|
| 28 |
+
"sliding_attention",
|
| 29 |
+
"sliding_attention",
|
| 30 |
+
"sliding_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"sliding_attention",
|
| 33 |
+
"sliding_attention",
|
| 34 |
+
"sliding_attention",
|
| 35 |
+
"sliding_attention",
|
| 36 |
+
"sliding_attention",
|
| 37 |
+
"full_attention"
|
| 38 |
+
],
|
| 39 |
+
"max_position_embeddings": 32768,
|
| 40 |
+
"model_type": "gemma3_text",
|
| 41 |
+
"neuron": {
|
| 42 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 43 |
+
"batch_size": 1,
|
| 44 |
+
"capacity_factor": null,
|
| 45 |
+
"checkpoint_id": "unsloth/gemma-3-270m-it",
|
| 46 |
+
"checkpoint_revision": "23cf460f6bb16954176b3ddcc8d4f250501458a9",
|
| 47 |
+
"continuous_batching": false,
|
| 48 |
+
"ep_degree": 1,
|
| 49 |
+
"fused_qkv": true,
|
| 50 |
+
"glu_mlp": true,
|
| 51 |
+
"local_ranks_size": 2,
|
| 52 |
+
"max_batch_size": 1,
|
| 53 |
+
"max_context_length": 8192,
|
| 54 |
+
"max_topk": 256,
|
| 55 |
+
"n_active_tokens": 8192,
|
| 56 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 57 |
+
"on_device_sampling": true,
|
| 58 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 59 |
+
"output_logits": false,
|
| 60 |
+
"pp_degree": 1,
|
| 61 |
+
"sequence_length": 8192,
|
| 62 |
+
"speculation_length": 0,
|
| 63 |
+
"start_rank_id": 0,
|
| 64 |
+
"target": "trn1",
|
| 65 |
+
"torch_dtype": "bfloat16",
|
| 66 |
+
"tp_degree": 2
|
| 67 |
+
},
|
| 68 |
+
"num_attention_heads": 4,
|
| 69 |
+
"num_hidden_layers": 18,
|
| 70 |
+
"num_key_value_heads": 1,
|
| 71 |
+
"query_pre_attn_scalar": 256,
|
| 72 |
+
"rms_norm_eps": 1e-06,
|
| 73 |
+
"rope_local_base_freq": 10000.0,
|
| 74 |
+
"rope_scaling": null,
|
| 75 |
+
"rope_theta": 1000000.0,
|
| 76 |
+
"sliding_window": 512,
|
| 77 |
+
"unsloth_fixed": true,
|
| 78 |
+
"use_bidirectional_attention": false,
|
| 79 |
+
"use_cache": true,
|
| 80 |
+
"vocab_size": 262144
|
| 81 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/granite/hf-internal-testing/tiny-random-GraniteForCausalLM/0e7a6f2933f99785cba6.json
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "hf-internal-testing/tiny-random-GraniteForCausalLM",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"GraniteForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"attention_multiplier": 1.0,
|
| 11 |
+
"dtype": "float32",
|
| 12 |
+
"embedding_multiplier": 1.0,
|
| 13 |
+
"hidden_act": "silu",
|
| 14 |
+
"hidden_size": 32,
|
| 15 |
+
"initializer_range": 0.02,
|
| 16 |
+
"intermediate_size": 64,
|
| 17 |
+
"logits_scaling": 1.0,
|
| 18 |
+
"max_position_embeddings": 2048,
|
| 19 |
+
"mlp_bias": false,
|
| 20 |
+
"model_type": "granite",
|
| 21 |
+
"neuron": {
|
| 22 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 23 |
+
"batch_size": 1,
|
| 24 |
+
"capacity_factor": null,
|
| 25 |
+
"checkpoint_id": "hf-internal-testing/tiny-random-GraniteForCausalLM",
|
| 26 |
+
"checkpoint_revision": "c3074ebc0ac2fe545305f5e5f6cce2cc9b2aa0c5",
|
| 27 |
+
"continuous_batching": false,
|
| 28 |
+
"ep_degree": 1,
|
| 29 |
+
"fused_qkv": true,
|
| 30 |
+
"glu_mlp": true,
|
| 31 |
+
"local_ranks_size": 2,
|
| 32 |
+
"max_batch_size": 1,
|
| 33 |
+
"max_context_length": 1024,
|
| 34 |
+
"max_topk": 256,
|
| 35 |
+
"n_active_tokens": 1024,
|
| 36 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 37 |
+
"on_device_sampling": true,
|
| 38 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 39 |
+
"output_logits": false,
|
| 40 |
+
"pp_degree": 1,
|
| 41 |
+
"sequence_length": 1024,
|
| 42 |
+
"speculation_length": 0,
|
| 43 |
+
"start_rank_id": 0,
|
| 44 |
+
"target": "trn1",
|
| 45 |
+
"torch_dtype": "float32",
|
| 46 |
+
"tp_degree": 2
|
| 47 |
+
},
|
| 48 |
+
"num_attention_heads": 4,
|
| 49 |
+
"num_hidden_layers": 2,
|
| 50 |
+
"num_key_value_heads": 4,
|
| 51 |
+
"residual_multiplier": 1.0,
|
| 52 |
+
"rms_norm_eps": 1e-06,
|
| 53 |
+
"rope_scaling": null,
|
| 54 |
+
"rope_theta": 10000.0,
|
| 55 |
+
"tie_word_embeddings": false,
|
| 56 |
+
"use_cache": true,
|
| 57 |
+
"vocab_size": 49152
|
| 58 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/granite/ibm-granite/granite-3.1-2b-instruct/efa7f046361aa22fb2b9.json
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "ibm-granite/granite-3.1-2b-instruct",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"GraniteForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.1,
|
| 10 |
+
"attention_multiplier": 0.015625,
|
| 11 |
+
"dtype": "bfloat16",
|
| 12 |
+
"embedding_multiplier": 12.0,
|
| 13 |
+
"hidden_act": "silu",
|
| 14 |
+
"hidden_size": 2048,
|
| 15 |
+
"initializer_range": 0.02,
|
| 16 |
+
"intermediate_size": 8192,
|
| 17 |
+
"logits_scaling": 8.0,
|
| 18 |
+
"max_position_embeddings": 131072,
|
| 19 |
+
"mlp_bias": false,
|
| 20 |
+
"model_type": "granite",
|
| 21 |
+
"neuron": {
|
| 22 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 23 |
+
"batch_size": 4,
|
| 24 |
+
"capacity_factor": null,
|
| 25 |
+
"checkpoint_id": "ibm-granite/granite-3.1-2b-instruct",
|
| 26 |
+
"checkpoint_revision": "bbc2aed595bd38bd770263dc3ab831db9794441d",
|
| 27 |
+
"continuous_batching": true,
|
| 28 |
+
"ep_degree": 1,
|
| 29 |
+
"fused_qkv": true,
|
| 30 |
+
"glu_mlp": true,
|
| 31 |
+
"local_ranks_size": 2,
|
| 32 |
+
"max_batch_size": 4,
|
| 33 |
+
"max_context_length": 4096,
|
| 34 |
+
"max_topk": 256,
|
| 35 |
+
"n_active_tokens": 4096,
|
| 36 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 37 |
+
"on_device_sampling": true,
|
| 38 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 39 |
+
"output_logits": false,
|
| 40 |
+
"pp_degree": 1,
|
| 41 |
+
"sequence_length": 4096,
|
| 42 |
+
"speculation_length": 0,
|
| 43 |
+
"start_rank_id": 0,
|
| 44 |
+
"target": "trn1",
|
| 45 |
+
"torch_dtype": "bfloat16",
|
| 46 |
+
"tp_degree": 2
|
| 47 |
+
},
|
| 48 |
+
"num_attention_heads": 32,
|
| 49 |
+
"num_hidden_layers": 40,
|
| 50 |
+
"num_key_value_heads": 8,
|
| 51 |
+
"residual_multiplier": 0.22,
|
| 52 |
+
"rms_norm_eps": 1e-05,
|
| 53 |
+
"rope_scaling": null,
|
| 54 |
+
"rope_theta": 5000000.0,
|
| 55 |
+
"tie_word_embeddings": true,
|
| 56 |
+
"use_cache": true,
|
| 57 |
+
"vocab_size": 49155
|
| 58 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/idefics3/HuggingFaceTB/SmolVLM-256M-Instruct/566b663bde0d89e24b29.json
ADDED
|
@@ -0,0 +1,188 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "HuggingFaceTB/SmolVLM-256M-Instruct",
|
| 4 |
+
"_task": "image-text-to-text",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Idefics3ForConditionalGeneration"
|
| 7 |
+
],
|
| 8 |
+
"dtype": "bfloat16",
|
| 9 |
+
"image_token_id": 49190,
|
| 10 |
+
"model_type": "idefics3",
|
| 11 |
+
"neuron": {
|
| 12 |
+
"_serialized_key": "NxDVLMNeuronConfig",
|
| 13 |
+
"batch_size": 1,
|
| 14 |
+
"capacity_factor": null,
|
| 15 |
+
"checkpoint_id": "HuggingFaceTB/SmolVLM-256M-Instruct",
|
| 16 |
+
"checkpoint_revision": "7e3e67edbbed1bf9888184d9df282b700a323964",
|
| 17 |
+
"continuous_batching": false,
|
| 18 |
+
"ep_degree": 1,
|
| 19 |
+
"fused_qkv": true,
|
| 20 |
+
"glu_mlp": true,
|
| 21 |
+
"image_seq_len": 64,
|
| 22 |
+
"image_size": 512,
|
| 23 |
+
"local_ranks_size": 2,
|
| 24 |
+
"max_batch_size": 1,
|
| 25 |
+
"max_context_length": 2048,
|
| 26 |
+
"max_num_images": 1,
|
| 27 |
+
"max_topk": 256,
|
| 28 |
+
"n_active_tokens": 2048,
|
| 29 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 30 |
+
"on_device_sampling": true,
|
| 31 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 32 |
+
"output_logits": false,
|
| 33 |
+
"pp_degree": 1,
|
| 34 |
+
"sequence_length": 2048,
|
| 35 |
+
"speculation_length": 0,
|
| 36 |
+
"start_rank_id": 0,
|
| 37 |
+
"target": "trn1",
|
| 38 |
+
"torch_dtype": "bfloat16",
|
| 39 |
+
"tp_degree": 2
|
| 40 |
+
},
|
| 41 |
+
"scale_factor": 4,
|
| 42 |
+
"text_config": {
|
| 43 |
+
"_attn_implementation_autoset": false,
|
| 44 |
+
"_flash_attn_2_enabled": true,
|
| 45 |
+
"_name_or_path": "None",
|
| 46 |
+
"architectures": [
|
| 47 |
+
"VLlama3ForCausalLM"
|
| 48 |
+
],
|
| 49 |
+
"attention_bias": false,
|
| 50 |
+
"attention_dropout": 0.0,
|
| 51 |
+
"dtype": "bfloat16",
|
| 52 |
+
"head_dim": 64,
|
| 53 |
+
"hidden_act": "silu",
|
| 54 |
+
"hidden_size": 576,
|
| 55 |
+
"initializer_range": 0.041666666666666664,
|
| 56 |
+
"intermediate_size": 1536,
|
| 57 |
+
"is_llama_config": true,
|
| 58 |
+
"max_position_embeddings": 8192,
|
| 59 |
+
"mlp_bias": false,
|
| 60 |
+
"model_type": "llama",
|
| 61 |
+
"neftune_noise_alpha": 0.0,
|
| 62 |
+
"num_attention_heads": 9,
|
| 63 |
+
"num_hidden_layers": 30,
|
| 64 |
+
"num_key_value_heads": 3,
|
| 65 |
+
"pad_token_id": 2,
|
| 66 |
+
"perceiver_config": {
|
| 67 |
+
"_attn_implementation_autoset": false,
|
| 68 |
+
"_name_or_path": "",
|
| 69 |
+
"add_cross_attention": false,
|
| 70 |
+
"architectures": null,
|
| 71 |
+
"attention_dropout": 0.0,
|
| 72 |
+
"bad_words_ids": null,
|
| 73 |
+
"begin_suppress_tokens": null,
|
| 74 |
+
"bos_token_id": null,
|
| 75 |
+
"chunk_size_feed_forward": 0,
|
| 76 |
+
"cross_attention_hidden_size": null,
|
| 77 |
+
"decoder_start_token_id": null,
|
| 78 |
+
"diversity_penalty": 0.0,
|
| 79 |
+
"do_sample": false,
|
| 80 |
+
"early_stopping": false,
|
| 81 |
+
"encoder_no_repeat_ngram_size": 0,
|
| 82 |
+
"eos_token_id": null,
|
| 83 |
+
"exponential_decay_length_penalty": null,
|
| 84 |
+
"finetuning_task": null,
|
| 85 |
+
"forced_bos_token_id": null,
|
| 86 |
+
"forced_eos_token_id": null,
|
| 87 |
+
"hidden_act": "silu",
|
| 88 |
+
"id2label": {
|
| 89 |
+
"0": "LABEL_0",
|
| 90 |
+
"1": "LABEL_1"
|
| 91 |
+
},
|
| 92 |
+
"is_decoder": false,
|
| 93 |
+
"is_encoder_decoder": false,
|
| 94 |
+
"label2id": {
|
| 95 |
+
"LABEL_0": 0,
|
| 96 |
+
"LABEL_1": 1
|
| 97 |
+
},
|
| 98 |
+
"length_penalty": 1.0,
|
| 99 |
+
"max_length": 20,
|
| 100 |
+
"min_length": 0,
|
| 101 |
+
"model_type": "vllama3",
|
| 102 |
+
"no_repeat_ngram_size": 0,
|
| 103 |
+
"num_beam_groups": 1,
|
| 104 |
+
"num_beams": 1,
|
| 105 |
+
"num_key_value_heads": 1,
|
| 106 |
+
"num_return_sequences": 1,
|
| 107 |
+
"output_attentions": false,
|
| 108 |
+
"output_hidden_states": false,
|
| 109 |
+
"output_scores": false,
|
| 110 |
+
"pad_token_id": null,
|
| 111 |
+
"prefix": null,
|
| 112 |
+
"problem_type": null,
|
| 113 |
+
"pruned_heads": {},
|
| 114 |
+
"qk_layer_norms_perceiver": false,
|
| 115 |
+
"remove_invalid_values": false,
|
| 116 |
+
"repetition_penalty": 1.0,
|
| 117 |
+
"resampler_depth": 6,
|
| 118 |
+
"resampler_head_dim": 96,
|
| 119 |
+
"resampler_n_heads": 16,
|
| 120 |
+
"resampler_n_latents": 64,
|
| 121 |
+
"return_dict": true,
|
| 122 |
+
"return_dict_in_generate": false,
|
| 123 |
+
"sep_token_id": null,
|
| 124 |
+
"suppress_tokens": null,
|
| 125 |
+
"task_specific_params": null,
|
| 126 |
+
"temperature": 1.0,
|
| 127 |
+
"tf_legacy_loss": false,
|
| 128 |
+
"tie_encoder_decoder": false,
|
| 129 |
+
"tie_word_embeddings": true,
|
| 130 |
+
"tokenizer_class": null,
|
| 131 |
+
"top_k": 50,
|
| 132 |
+
"top_p": 1.0,
|
| 133 |
+
"torch_dtype": null,
|
| 134 |
+
"torchscript": false,
|
| 135 |
+
"transformers_version": "4.46.0",
|
| 136 |
+
"typical_p": 1.0,
|
| 137 |
+
"use_bfloat16": false
|
| 138 |
+
},
|
| 139 |
+
"pixel_shuffle_factor": 4,
|
| 140 |
+
"pretraining_tp": 1,
|
| 141 |
+
"qk_layer_norms": false,
|
| 142 |
+
"rms_norm_eps": 1e-05,
|
| 143 |
+
"rope_interleaved": false,
|
| 144 |
+
"rope_scaling": null,
|
| 145 |
+
"rope_theta": 100000,
|
| 146 |
+
"transformers.js_config": {
|
| 147 |
+
"kv_cache_dtype": {
|
| 148 |
+
"fp16": "float16",
|
| 149 |
+
"q4f16": "float16"
|
| 150 |
+
}
|
| 151 |
+
},
|
| 152 |
+
"use_cache": true,
|
| 153 |
+
"use_resampler": false,
|
| 154 |
+
"vocab_size": 49280
|
| 155 |
+
},
|
| 156 |
+
"tie_word_embeddings": false,
|
| 157 |
+
"transformers.js_config": {
|
| 158 |
+
"kv_cache_dtype": {
|
| 159 |
+
"fp16": "float16",
|
| 160 |
+
"q4f16": "float16"
|
| 161 |
+
}
|
| 162 |
+
},
|
| 163 |
+
"use_cache": true,
|
| 164 |
+
"vision_config": {
|
| 165 |
+
"_attn_implementation_autoset": false,
|
| 166 |
+
"attention_dropout": 0.0,
|
| 167 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 168 |
+
"hidden_size": 768,
|
| 169 |
+
"image_size": 512,
|
| 170 |
+
"initializer_range": 0.02,
|
| 171 |
+
"intermediate_size": 3072,
|
| 172 |
+
"layer_norm_eps": 1e-06,
|
| 173 |
+
"max_image_size": {
|
| 174 |
+
"longest_edge": 512
|
| 175 |
+
},
|
| 176 |
+
"model_type": "idefics3_vision",
|
| 177 |
+
"num_attention_heads": 12,
|
| 178 |
+
"num_channels": 3,
|
| 179 |
+
"num_hidden_layers": 12,
|
| 180 |
+
"patch_size": 16,
|
| 181 |
+
"size": {
|
| 182 |
+
"longest_edge": 2048
|
| 183 |
+
},
|
| 184 |
+
"tie_word_embeddings": false,
|
| 185 |
+
"use_base_siglip": true
|
| 186 |
+
},
|
| 187 |
+
"vocab_size": 49280
|
| 188 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/llama/llamafactory/tiny-random-Llama-3/97e03fcfc7f46ac8836e.json
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "llamafactory/tiny-random-Llama-3",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"LlamaForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "float16",
|
| 11 |
+
"head_dim": 4,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 16,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 64,
|
| 16 |
+
"max_position_embeddings": 131072,
|
| 17 |
+
"mlp_bias": false,
|
| 18 |
+
"model_type": "llama",
|
| 19 |
+
"neuron": {
|
| 20 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 21 |
+
"batch_size": 1,
|
| 22 |
+
"capacity_factor": null,
|
| 23 |
+
"checkpoint_id": "llamafactory/tiny-random-Llama-3",
|
| 24 |
+
"checkpoint_revision": "bf2a2e3bf199ad2ee96f02a3c00246c608db22a8",
|
| 25 |
+
"continuous_batching": false,
|
| 26 |
+
"ep_degree": 1,
|
| 27 |
+
"fused_qkv": true,
|
| 28 |
+
"glu_mlp": true,
|
| 29 |
+
"local_ranks_size": 2,
|
| 30 |
+
"max_batch_size": 1,
|
| 31 |
+
"max_context_length": 1024,
|
| 32 |
+
"max_topk": 256,
|
| 33 |
+
"n_active_tokens": 1024,
|
| 34 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 35 |
+
"on_device_sampling": true,
|
| 36 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 37 |
+
"output_logits": false,
|
| 38 |
+
"pp_degree": 1,
|
| 39 |
+
"sequence_length": 1024,
|
| 40 |
+
"speculation_length": 0,
|
| 41 |
+
"start_rank_id": 0,
|
| 42 |
+
"target": "trn1",
|
| 43 |
+
"torch_dtype": "float16",
|
| 44 |
+
"tp_degree": 2
|
| 45 |
+
},
|
| 46 |
+
"num_attention_heads": 4,
|
| 47 |
+
"num_hidden_layers": 2,
|
| 48 |
+
"num_key_value_heads": 4,
|
| 49 |
+
"pretraining_tp": 1,
|
| 50 |
+
"rms_norm_eps": 1e-05,
|
| 51 |
+
"rope_scaling": {
|
| 52 |
+
"factor": 8.0,
|
| 53 |
+
"high_freq_factor": 4.0,
|
| 54 |
+
"low_freq_factor": 1.0,
|
| 55 |
+
"original_max_position_embeddings": 8192,
|
| 56 |
+
"rope_type": "llama3"
|
| 57 |
+
},
|
| 58 |
+
"rope_theta": 500000.0,
|
| 59 |
+
"tie_word_embeddings": false,
|
| 60 |
+
"use_cache": true,
|
| 61 |
+
"vocab_size": 128256
|
| 62 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/llama/unsloth/Llama-3.2-1B-Instruct/4d99c6a74830655285f6.json
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "unsloth/Llama-3.2-1B-Instruct",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"LlamaForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 64,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 2048,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 8192,
|
| 16 |
+
"max_position_embeddings": 131072,
|
| 17 |
+
"mlp_bias": false,
|
| 18 |
+
"model_type": "llama",
|
| 19 |
+
"neuron": {
|
| 20 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 21 |
+
"batch_size": 4,
|
| 22 |
+
"capacity_factor": null,
|
| 23 |
+
"checkpoint_id": "unsloth/Llama-3.2-1B-Instruct",
|
| 24 |
+
"checkpoint_revision": "5a8abab4a5d6f164389b1079fb721cfab8d7126c",
|
| 25 |
+
"continuous_batching": true,
|
| 26 |
+
"ep_degree": 1,
|
| 27 |
+
"fused_qkv": true,
|
| 28 |
+
"glu_mlp": true,
|
| 29 |
+
"local_ranks_size": 2,
|
| 30 |
+
"max_batch_size": 4,
|
| 31 |
+
"max_context_length": 4096,
|
| 32 |
+
"max_topk": 256,
|
| 33 |
+
"n_active_tokens": 4096,
|
| 34 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 35 |
+
"on_device_sampling": true,
|
| 36 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 37 |
+
"output_logits": false,
|
| 38 |
+
"pp_degree": 1,
|
| 39 |
+
"sequence_length": 4096,
|
| 40 |
+
"speculation_length": 0,
|
| 41 |
+
"start_rank_id": 0,
|
| 42 |
+
"target": "trn1",
|
| 43 |
+
"torch_dtype": "bfloat16",
|
| 44 |
+
"tp_degree": 2
|
| 45 |
+
},
|
| 46 |
+
"num_attention_heads": 32,
|
| 47 |
+
"num_hidden_layers": 16,
|
| 48 |
+
"num_key_value_heads": 8,
|
| 49 |
+
"pretraining_tp": 1,
|
| 50 |
+
"rms_norm_eps": 1e-05,
|
| 51 |
+
"rope_scaling": {
|
| 52 |
+
"factor": 32.0,
|
| 53 |
+
"high_freq_factor": 4.0,
|
| 54 |
+
"low_freq_factor": 1.0,
|
| 55 |
+
"original_max_position_embeddings": 8192,
|
| 56 |
+
"rope_type": "llama3"
|
| 57 |
+
},
|
| 58 |
+
"rope_theta": 500000.0,
|
| 59 |
+
"tie_word_embeddings": true,
|
| 60 |
+
"unsloth_fixed": true,
|
| 61 |
+
"use_cache": true,
|
| 62 |
+
"vocab_size": 128256
|
| 63 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/llama4/tiny-random/llama-4/dadf01a9f544218eabdd.json
ADDED
|
@@ -0,0 +1,125 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "tiny-random/llama-4",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Llama4ForConditionalGeneration"
|
| 7 |
+
],
|
| 8 |
+
"boi_token_index": 200080,
|
| 9 |
+
"dtype": "bfloat16",
|
| 10 |
+
"eoi_token_index": 200081,
|
| 11 |
+
"image_token_index": 200092,
|
| 12 |
+
"model_type": "llama4",
|
| 13 |
+
"neuron": {
|
| 14 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 15 |
+
"batch_size": 1,
|
| 16 |
+
"capacity_factor": null,
|
| 17 |
+
"checkpoint_id": "tiny-random/llama-4",
|
| 18 |
+
"checkpoint_revision": "9e716f5d4d1ffe0a44a15f46f4a12b840439aba4",
|
| 19 |
+
"continuous_batching": false,
|
| 20 |
+
"ep_degree": 1,
|
| 21 |
+
"fused_qkv": false,
|
| 22 |
+
"glu_mlp": true,
|
| 23 |
+
"local_ranks_size": 2,
|
| 24 |
+
"max_batch_size": 1,
|
| 25 |
+
"max_context_length": 1024,
|
| 26 |
+
"max_topk": 256,
|
| 27 |
+
"n_active_tokens": 1024,
|
| 28 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 29 |
+
"on_device_sampling": true,
|
| 30 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 31 |
+
"output_logits": false,
|
| 32 |
+
"pp_degree": 1,
|
| 33 |
+
"sequence_length": 1024,
|
| 34 |
+
"speculation_length": 0,
|
| 35 |
+
"start_rank_id": 0,
|
| 36 |
+
"target": "trn1",
|
| 37 |
+
"torch_dtype": "bfloat16",
|
| 38 |
+
"tp_degree": 2
|
| 39 |
+
},
|
| 40 |
+
"text_config": {
|
| 41 |
+
"_attn_implementation_autoset": true,
|
| 42 |
+
"attention_bias": false,
|
| 43 |
+
"attention_chunk_size": 128,
|
| 44 |
+
"attention_dropout": 0.0,
|
| 45 |
+
"attn_scale": 0.1,
|
| 46 |
+
"attn_temperature_tuning": 4,
|
| 47 |
+
"bos_token_id": 200000,
|
| 48 |
+
"cache_implementation": "hybrid",
|
| 49 |
+
"dtype": "bfloat16",
|
| 50 |
+
"eos_token_id": [
|
| 51 |
+
200001,
|
| 52 |
+
200007,
|
| 53 |
+
200008
|
| 54 |
+
],
|
| 55 |
+
"floor_scale": 8192,
|
| 56 |
+
"for_llm_compressor": false,
|
| 57 |
+
"head_dim": 32,
|
| 58 |
+
"hidden_act": "silu",
|
| 59 |
+
"hidden_size": 32,
|
| 60 |
+
"initializer_range": 0.02,
|
| 61 |
+
"interleave_moe_layer_step": 2,
|
| 62 |
+
"intermediate_size": 64,
|
| 63 |
+
"intermediate_size_mlp": 128,
|
| 64 |
+
"layer_types": [
|
| 65 |
+
"chunked_attention",
|
| 66 |
+
"chunked_attention",
|
| 67 |
+
"chunked_attention",
|
| 68 |
+
"full_attention"
|
| 69 |
+
],
|
| 70 |
+
"max_position_embeddings": 1048576,
|
| 71 |
+
"model_type": "llama4_text",
|
| 72 |
+
"moe_layers": [
|
| 73 |
+
1,
|
| 74 |
+
3
|
| 75 |
+
],
|
| 76 |
+
"no_rope_layers": [
|
| 77 |
+
1,
|
| 78 |
+
1,
|
| 79 |
+
1,
|
| 80 |
+
0
|
| 81 |
+
],
|
| 82 |
+
"num_attention_heads": 1,
|
| 83 |
+
"num_experts_per_tok": 1,
|
| 84 |
+
"num_hidden_layers": 4,
|
| 85 |
+
"num_key_value_heads": 1,
|
| 86 |
+
"num_local_experts": 8,
|
| 87 |
+
"output_router_logits": false,
|
| 88 |
+
"pad_token_id": 200018,
|
| 89 |
+
"rms_norm_eps": 1e-05,
|
| 90 |
+
"rope_scaling": null,
|
| 91 |
+
"rope_theta": 500000.0,
|
| 92 |
+
"router_aux_loss_coef": 0.001,
|
| 93 |
+
"router_jitter_noise": 0.0,
|
| 94 |
+
"tie_word_embeddings": true,
|
| 95 |
+
"use_cache": true,
|
| 96 |
+
"use_qk_norm": true,
|
| 97 |
+
"vocab_size": 202048
|
| 98 |
+
},
|
| 99 |
+
"tie_word_embeddings": false,
|
| 100 |
+
"vision_config": {
|
| 101 |
+
"_attn_implementation_autoset": true,
|
| 102 |
+
"_vision_feature_layer": -1,
|
| 103 |
+
"attention_dropout": 0.0,
|
| 104 |
+
"hidden_act": "gelu",
|
| 105 |
+
"hidden_size": 32,
|
| 106 |
+
"image_size": 336,
|
| 107 |
+
"initializer_range": 0.02,
|
| 108 |
+
"intermediate_size": 128,
|
| 109 |
+
"model_type": "llama4_vision_model",
|
| 110 |
+
"multi_modal_projector_bias": false,
|
| 111 |
+
"norm_eps": 1e-05,
|
| 112 |
+
"num_attention_heads": 1,
|
| 113 |
+
"num_channels": 3,
|
| 114 |
+
"num_hidden_layers": 2,
|
| 115 |
+
"patch_size": 14,
|
| 116 |
+
"pixel_shuffle_ratio": 0.5,
|
| 117 |
+
"projector_dropout": 0.0,
|
| 118 |
+
"projector_input_dim": 32,
|
| 119 |
+
"projector_output_dim": 32,
|
| 120 |
+
"rope_theta": 10000,
|
| 121 |
+
"vision_feature_layer": -1,
|
| 122 |
+
"vision_feature_select_strategy": "default",
|
| 123 |
+
"vision_output_dim": 32
|
| 124 |
+
}
|
| 125 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/mixtral/dacorvo/Mixtral-tiny/03a64a22d1b885eece61.json
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "dacorvo/Mixtral-tiny",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"MixtralForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_dropout": 0.0,
|
| 9 |
+
"dtype": "float16",
|
| 10 |
+
"head_dim": 32,
|
| 11 |
+
"hidden_act": "silu",
|
| 12 |
+
"hidden_size": 1024,
|
| 13 |
+
"initializer_range": 0.02,
|
| 14 |
+
"intermediate_size": 3584,
|
| 15 |
+
"max_position_embeddings": 1024,
|
| 16 |
+
"model_type": "mixtral",
|
| 17 |
+
"neuron": {
|
| 18 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 19 |
+
"batch_size": 1,
|
| 20 |
+
"capacity_factor": null,
|
| 21 |
+
"checkpoint_id": "dacorvo/Mixtral-tiny",
|
| 22 |
+
"checkpoint_revision": "c557ba205ddff6ea911f4719e0d543d6c08356b6",
|
| 23 |
+
"continuous_batching": false,
|
| 24 |
+
"ep_degree": 1,
|
| 25 |
+
"fused_qkv": false,
|
| 26 |
+
"glu_mlp": true,
|
| 27 |
+
"local_ranks_size": 2,
|
| 28 |
+
"max_batch_size": 1,
|
| 29 |
+
"max_context_length": 1024,
|
| 30 |
+
"max_topk": 256,
|
| 31 |
+
"n_active_tokens": 1024,
|
| 32 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 33 |
+
"on_device_sampling": false,
|
| 34 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 35 |
+
"output_logits": false,
|
| 36 |
+
"pp_degree": 1,
|
| 37 |
+
"sequence_length": 1024,
|
| 38 |
+
"speculation_length": 0,
|
| 39 |
+
"start_rank_id": 0,
|
| 40 |
+
"target": "trn1",
|
| 41 |
+
"torch_dtype": "float16",
|
| 42 |
+
"tp_degree": 2
|
| 43 |
+
},
|
| 44 |
+
"num_attention_heads": 32,
|
| 45 |
+
"num_experts_per_tok": 2,
|
| 46 |
+
"num_hidden_layers": 2,
|
| 47 |
+
"num_key_value_heads": 8,
|
| 48 |
+
"num_local_experts": 8,
|
| 49 |
+
"output_router_logits": false,
|
| 50 |
+
"rms_norm_eps": 1e-05,
|
| 51 |
+
"rope_theta": 10000.0,
|
| 52 |
+
"router_aux_loss_coef": 0.001,
|
| 53 |
+
"router_jitter_noise": 0.0,
|
| 54 |
+
"sliding_window": 4096,
|
| 55 |
+
"tie_word_embeddings": false,
|
| 56 |
+
"use_cache": true,
|
| 57 |
+
"vocab_size": 32000
|
| 58 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/phi3/microsoft/Phi-3.5-mini-instruct/d00ca5f7300e7c2696dd.json
ADDED
|
@@ -0,0 +1,164 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "microsoft/Phi-3.5-mini-instruct",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Phi3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"auto_map": {
|
| 11 |
+
"AutoConfig": "configuration_phi3.Phi3Config",
|
| 12 |
+
"AutoModelForCausalLM": "modeling_phi3.Phi3ForCausalLM"
|
| 13 |
+
},
|
| 14 |
+
"dtype": "bfloat16",
|
| 15 |
+
"embd_pdrop": 0.0,
|
| 16 |
+
"hidden_act": "silu",
|
| 17 |
+
"hidden_size": 3072,
|
| 18 |
+
"initializer_range": 0.02,
|
| 19 |
+
"intermediate_size": 8192,
|
| 20 |
+
"max_position_embeddings": 131072,
|
| 21 |
+
"model_type": "phi3",
|
| 22 |
+
"neuron": {
|
| 23 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 24 |
+
"batch_size": 4,
|
| 25 |
+
"capacity_factor": null,
|
| 26 |
+
"checkpoint_id": "microsoft/Phi-3.5-mini-instruct",
|
| 27 |
+
"checkpoint_revision": "2fe192450127e6a83f7441aef6e3ca586c338b77",
|
| 28 |
+
"continuous_batching": true,
|
| 29 |
+
"ep_degree": 1,
|
| 30 |
+
"fused_qkv": true,
|
| 31 |
+
"glu_mlp": true,
|
| 32 |
+
"local_ranks_size": 2,
|
| 33 |
+
"max_batch_size": 4,
|
| 34 |
+
"max_context_length": 4096,
|
| 35 |
+
"max_topk": 256,
|
| 36 |
+
"n_active_tokens": 4096,
|
| 37 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 38 |
+
"on_device_sampling": true,
|
| 39 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 40 |
+
"output_logits": false,
|
| 41 |
+
"pp_degree": 1,
|
| 42 |
+
"sequence_length": 4096,
|
| 43 |
+
"speculation_length": 0,
|
| 44 |
+
"start_rank_id": 0,
|
| 45 |
+
"target": "trn1",
|
| 46 |
+
"torch_dtype": "bfloat16",
|
| 47 |
+
"tp_degree": 2
|
| 48 |
+
},
|
| 49 |
+
"num_attention_heads": 32,
|
| 50 |
+
"num_hidden_layers": 32,
|
| 51 |
+
"num_key_value_heads": 32,
|
| 52 |
+
"original_max_position_embeddings": 4096,
|
| 53 |
+
"partial_rotary_factor": 1.0,
|
| 54 |
+
"resid_pdrop": 0.0,
|
| 55 |
+
"rms_norm_eps": 1e-05,
|
| 56 |
+
"rope_scaling": {
|
| 57 |
+
"long_factor": [
|
| 58 |
+
1.0800000429153442,
|
| 59 |
+
1.1100000143051147,
|
| 60 |
+
1.1399999856948853,
|
| 61 |
+
1.340000033378601,
|
| 62 |
+
1.5899999141693115,
|
| 63 |
+
1.600000023841858,
|
| 64 |
+
1.6200000047683716,
|
| 65 |
+
2.620000123977661,
|
| 66 |
+
3.2300000190734863,
|
| 67 |
+
3.2300000190734863,
|
| 68 |
+
4.789999961853027,
|
| 69 |
+
7.400000095367432,
|
| 70 |
+
7.700000286102295,
|
| 71 |
+
9.09000015258789,
|
| 72 |
+
12.199999809265137,
|
| 73 |
+
17.670000076293945,
|
| 74 |
+
24.46000099182129,
|
| 75 |
+
28.57000160217285,
|
| 76 |
+
30.420001983642578,
|
| 77 |
+
30.840002059936523,
|
| 78 |
+
32.590003967285156,
|
| 79 |
+
32.93000411987305,
|
| 80 |
+
42.320003509521484,
|
| 81 |
+
44.96000289916992,
|
| 82 |
+
50.340003967285156,
|
| 83 |
+
50.45000457763672,
|
| 84 |
+
57.55000305175781,
|
| 85 |
+
57.93000411987305,
|
| 86 |
+
58.21000289916992,
|
| 87 |
+
60.1400032043457,
|
| 88 |
+
62.61000442504883,
|
| 89 |
+
62.62000274658203,
|
| 90 |
+
62.71000289916992,
|
| 91 |
+
63.1400032043457,
|
| 92 |
+
63.1400032043457,
|
| 93 |
+
63.77000427246094,
|
| 94 |
+
63.93000411987305,
|
| 95 |
+
63.96000289916992,
|
| 96 |
+
63.970001220703125,
|
| 97 |
+
64.02999877929688,
|
| 98 |
+
64.06999969482422,
|
| 99 |
+
64.08000183105469,
|
| 100 |
+
64.12000274658203,
|
| 101 |
+
64.41000366210938,
|
| 102 |
+
64.4800033569336,
|
| 103 |
+
64.51000213623047,
|
| 104 |
+
64.52999877929688,
|
| 105 |
+
64.83999633789062
|
| 106 |
+
],
|
| 107 |
+
"short_factor": [
|
| 108 |
+
1.0,
|
| 109 |
+
1.0199999809265137,
|
| 110 |
+
1.0299999713897705,
|
| 111 |
+
1.0299999713897705,
|
| 112 |
+
1.0499999523162842,
|
| 113 |
+
1.0499999523162842,
|
| 114 |
+
1.0499999523162842,
|
| 115 |
+
1.0499999523162842,
|
| 116 |
+
1.0499999523162842,
|
| 117 |
+
1.0699999332427979,
|
| 118 |
+
1.0999999046325684,
|
| 119 |
+
1.1099998950958252,
|
| 120 |
+
1.1599998474121094,
|
| 121 |
+
1.1599998474121094,
|
| 122 |
+
1.1699998378753662,
|
| 123 |
+
1.2899998426437378,
|
| 124 |
+
1.339999794960022,
|
| 125 |
+
1.679999828338623,
|
| 126 |
+
1.7899998426437378,
|
| 127 |
+
1.8199998140335083,
|
| 128 |
+
1.8499997854232788,
|
| 129 |
+
1.8799997568130493,
|
| 130 |
+
1.9099997282028198,
|
| 131 |
+
1.9399996995925903,
|
| 132 |
+
1.9899996519088745,
|
| 133 |
+
2.0199997425079346,
|
| 134 |
+
2.0199997425079346,
|
| 135 |
+
2.0199997425079346,
|
| 136 |
+
2.0199997425079346,
|
| 137 |
+
2.0199997425079346,
|
| 138 |
+
2.0199997425079346,
|
| 139 |
+
2.0299997329711914,
|
| 140 |
+
2.0299997329711914,
|
| 141 |
+
2.0299997329711914,
|
| 142 |
+
2.0299997329711914,
|
| 143 |
+
2.0299997329711914,
|
| 144 |
+
2.0299997329711914,
|
| 145 |
+
2.0299997329711914,
|
| 146 |
+
2.0299997329711914,
|
| 147 |
+
2.0299997329711914,
|
| 148 |
+
2.0799996852874756,
|
| 149 |
+
2.0899996757507324,
|
| 150 |
+
2.189999580383301,
|
| 151 |
+
2.2199995517730713,
|
| 152 |
+
2.5899994373321533,
|
| 153 |
+
2.729999542236328,
|
| 154 |
+
2.749999523162842,
|
| 155 |
+
2.8399994373321533
|
| 156 |
+
],
|
| 157 |
+
"type": "longrope"
|
| 158 |
+
},
|
| 159 |
+
"rope_theta": 10000.0,
|
| 160 |
+
"sliding_window": 262144,
|
| 161 |
+
"tie_word_embeddings": false,
|
| 162 |
+
"use_cache": true,
|
| 163 |
+
"vocab_size": 32064
|
| 164 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/phi3/yujiepan/phi-4-tiny-random/e5ddd3afb102c02baf93.json
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "yujiepan/phi-4-tiny-random",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Phi3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"auto_map": {},
|
| 11 |
+
"dtype": "bfloat16",
|
| 12 |
+
"embd_pdrop": 0.0,
|
| 13 |
+
"hidden_act": "silu",
|
| 14 |
+
"hidden_size": 16,
|
| 15 |
+
"initializer_range": 0.02,
|
| 16 |
+
"intermediate_size": 32,
|
| 17 |
+
"max_position_embeddings": 16384,
|
| 18 |
+
"model_type": "phi3",
|
| 19 |
+
"neuron": {
|
| 20 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 21 |
+
"batch_size": 1,
|
| 22 |
+
"capacity_factor": null,
|
| 23 |
+
"checkpoint_id": "yujiepan/phi-4-tiny-random",
|
| 24 |
+
"checkpoint_revision": "18a9a1168dc97ac6d128f811925670c275610f5a",
|
| 25 |
+
"continuous_batching": false,
|
| 26 |
+
"ep_degree": 1,
|
| 27 |
+
"fused_qkv": true,
|
| 28 |
+
"glu_mlp": true,
|
| 29 |
+
"local_ranks_size": 2,
|
| 30 |
+
"max_batch_size": 1,
|
| 31 |
+
"max_context_length": 1024,
|
| 32 |
+
"max_topk": 256,
|
| 33 |
+
"n_active_tokens": 1024,
|
| 34 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 35 |
+
"on_device_sampling": true,
|
| 36 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 37 |
+
"output_logits": false,
|
| 38 |
+
"pp_degree": 1,
|
| 39 |
+
"sequence_length": 1024,
|
| 40 |
+
"speculation_length": 0,
|
| 41 |
+
"start_rank_id": 0,
|
| 42 |
+
"target": "trn1",
|
| 43 |
+
"torch_dtype": "bfloat16",
|
| 44 |
+
"tp_degree": 2
|
| 45 |
+
},
|
| 46 |
+
"num_attention_heads": 2,
|
| 47 |
+
"num_hidden_layers": 2,
|
| 48 |
+
"num_key_value_heads": 1,
|
| 49 |
+
"original_max_position_embeddings": 16384,
|
| 50 |
+
"partial_rotary_factor": 1.0,
|
| 51 |
+
"resid_pdrop": 0.0,
|
| 52 |
+
"rms_norm_eps": 1e-05,
|
| 53 |
+
"rope_scaling": null,
|
| 54 |
+
"rope_theta": 250000,
|
| 55 |
+
"sliding_window": null,
|
| 56 |
+
"tie_word_embeddings": false,
|
| 57 |
+
"use_cache": true,
|
| 58 |
+
"vocab_size": 100352
|
| 59 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/qwen2/Qwen/Qwen2.5-0.5B/1c8b4a21eb41ff235945.json
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen2.5-0.5B",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen2ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_dropout": 0.0,
|
| 9 |
+
"dtype": "bfloat16",
|
| 10 |
+
"hidden_act": "silu",
|
| 11 |
+
"hidden_size": 896,
|
| 12 |
+
"initializer_range": 0.02,
|
| 13 |
+
"intermediate_size": 4864,
|
| 14 |
+
"layer_types": [
|
| 15 |
+
"full_attention",
|
| 16 |
+
"full_attention",
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention"
|
| 39 |
+
],
|
| 40 |
+
"max_position_embeddings": 32768,
|
| 41 |
+
"max_window_layers": 24,
|
| 42 |
+
"model_type": "qwen2",
|
| 43 |
+
"neuron": {
|
| 44 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 45 |
+
"batch_size": 4,
|
| 46 |
+
"capacity_factor": null,
|
| 47 |
+
"checkpoint_id": "Qwen/Qwen2.5-0.5B",
|
| 48 |
+
"checkpoint_revision": "060db6499f32faf8b98477b0a26969ef7d8b9987",
|
| 49 |
+
"continuous_batching": true,
|
| 50 |
+
"ep_degree": 1,
|
| 51 |
+
"fused_qkv": false,
|
| 52 |
+
"glu_mlp": true,
|
| 53 |
+
"local_ranks_size": 2,
|
| 54 |
+
"max_batch_size": 4,
|
| 55 |
+
"max_context_length": 4096,
|
| 56 |
+
"max_topk": 256,
|
| 57 |
+
"n_active_tokens": 4096,
|
| 58 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 59 |
+
"on_device_sampling": false,
|
| 60 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 61 |
+
"output_logits": false,
|
| 62 |
+
"pp_degree": 1,
|
| 63 |
+
"sequence_length": 4096,
|
| 64 |
+
"speculation_length": 0,
|
| 65 |
+
"start_rank_id": 0,
|
| 66 |
+
"target": "trn1",
|
| 67 |
+
"torch_dtype": "bfloat16",
|
| 68 |
+
"tp_degree": 2
|
| 69 |
+
},
|
| 70 |
+
"num_attention_heads": 14,
|
| 71 |
+
"num_hidden_layers": 24,
|
| 72 |
+
"num_key_value_heads": 2,
|
| 73 |
+
"rms_norm_eps": 1e-06,
|
| 74 |
+
"rope_scaling": null,
|
| 75 |
+
"rope_theta": 1000000.0,
|
| 76 |
+
"sliding_window": null,
|
| 77 |
+
"tie_word_embeddings": true,
|
| 78 |
+
"use_cache": true,
|
| 79 |
+
"use_mrope": false,
|
| 80 |
+
"use_sliding_window": false,
|
| 81 |
+
"vocab_size": 151936
|
| 82 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/qwen2/yujiepan/qwen2.5-128k-tiny-random/29c61ad2f54baaec4c1d.json
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "yujiepan/qwen2.5-128k-tiny-random",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen2ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_dropout": 0.0,
|
| 9 |
+
"dtype": "bfloat16",
|
| 10 |
+
"hidden_act": "silu",
|
| 11 |
+
"hidden_size": 8,
|
| 12 |
+
"initializer_range": 0.02,
|
| 13 |
+
"intermediate_size": 16,
|
| 14 |
+
"layer_types": [
|
| 15 |
+
"full_attention",
|
| 16 |
+
"full_attention"
|
| 17 |
+
],
|
| 18 |
+
"max_position_embeddings": 32768,
|
| 19 |
+
"max_window_layers": 1,
|
| 20 |
+
"model_type": "qwen2",
|
| 21 |
+
"neuron": {
|
| 22 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 23 |
+
"batch_size": 1,
|
| 24 |
+
"capacity_factor": null,
|
| 25 |
+
"checkpoint_id": "yujiepan/qwen2.5-128k-tiny-random",
|
| 26 |
+
"checkpoint_revision": "c8296d4ca3f87782876d2382fbb6481d1beb8ef0",
|
| 27 |
+
"continuous_batching": false,
|
| 28 |
+
"ep_degree": 1,
|
| 29 |
+
"fused_qkv": false,
|
| 30 |
+
"glu_mlp": true,
|
| 31 |
+
"local_ranks_size": 2,
|
| 32 |
+
"max_batch_size": 1,
|
| 33 |
+
"max_context_length": 1024,
|
| 34 |
+
"max_topk": 256,
|
| 35 |
+
"n_active_tokens": 1024,
|
| 36 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 37 |
+
"on_device_sampling": true,
|
| 38 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 39 |
+
"output_logits": false,
|
| 40 |
+
"pp_degree": 1,
|
| 41 |
+
"sequence_length": 1024,
|
| 42 |
+
"speculation_length": 0,
|
| 43 |
+
"start_rank_id": 0,
|
| 44 |
+
"target": "trn1",
|
| 45 |
+
"torch_dtype": "bfloat16",
|
| 46 |
+
"tp_degree": 2
|
| 47 |
+
},
|
| 48 |
+
"num_attention_heads": 4,
|
| 49 |
+
"num_hidden_layers": 2,
|
| 50 |
+
"num_key_value_heads": 2,
|
| 51 |
+
"rms_norm_eps": 1e-06,
|
| 52 |
+
"rope_scaling": {
|
| 53 |
+
"factor": 4.0,
|
| 54 |
+
"original_max_position_embeddings": 32768,
|
| 55 |
+
"rope_type": "yarn",
|
| 56 |
+
"type": "yarn"
|
| 57 |
+
},
|
| 58 |
+
"rope_theta": 1000000.0,
|
| 59 |
+
"sliding_window": null,
|
| 60 |
+
"tie_word_embeddings": false,
|
| 61 |
+
"use_cache": true,
|
| 62 |
+
"use_sliding_window": false,
|
| 63 |
+
"vocab_size": 152064
|
| 64 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/qwen3/Qwen/Qwen3-0.6B/f65f144a780153ddd757.json
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "Qwen/Qwen3-0.6B",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"dtype": "bfloat16",
|
| 11 |
+
"head_dim": 128,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 1024,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 3072,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"full_attention",
|
| 18 |
+
"full_attention",
|
| 19 |
+
"full_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"full_attention",
|
| 22 |
+
"full_attention",
|
| 23 |
+
"full_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"full_attention",
|
| 26 |
+
"full_attention",
|
| 27 |
+
"full_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"full_attention",
|
| 30 |
+
"full_attention",
|
| 31 |
+
"full_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"full_attention",
|
| 34 |
+
"full_attention",
|
| 35 |
+
"full_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"full_attention",
|
| 38 |
+
"full_attention",
|
| 39 |
+
"full_attention",
|
| 40 |
+
"full_attention",
|
| 41 |
+
"full_attention",
|
| 42 |
+
"full_attention",
|
| 43 |
+
"full_attention",
|
| 44 |
+
"full_attention"
|
| 45 |
+
],
|
| 46 |
+
"max_position_embeddings": 40960,
|
| 47 |
+
"max_window_layers": 28,
|
| 48 |
+
"model_type": "qwen3",
|
| 49 |
+
"neuron": {
|
| 50 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 51 |
+
"batch_size": 4,
|
| 52 |
+
"capacity_factor": null,
|
| 53 |
+
"checkpoint_id": "Qwen/Qwen3-0.6B",
|
| 54 |
+
"checkpoint_revision": "c1899de289a04d12100db370d81485cdf75e47ca",
|
| 55 |
+
"continuous_batching": true,
|
| 56 |
+
"ep_degree": 1,
|
| 57 |
+
"fused_qkv": true,
|
| 58 |
+
"glu_mlp": true,
|
| 59 |
+
"local_ranks_size": 2,
|
| 60 |
+
"max_batch_size": 4,
|
| 61 |
+
"max_context_length": 4096,
|
| 62 |
+
"max_topk": 256,
|
| 63 |
+
"n_active_tokens": 4096,
|
| 64 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 65 |
+
"on_device_sampling": false,
|
| 66 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 67 |
+
"output_logits": false,
|
| 68 |
+
"pp_degree": 1,
|
| 69 |
+
"sequence_length": 4096,
|
| 70 |
+
"speculation_length": 0,
|
| 71 |
+
"start_rank_id": 0,
|
| 72 |
+
"target": "trn1",
|
| 73 |
+
"torch_dtype": "bfloat16",
|
| 74 |
+
"tp_degree": 2
|
| 75 |
+
},
|
| 76 |
+
"num_attention_heads": 16,
|
| 77 |
+
"num_hidden_layers": 28,
|
| 78 |
+
"num_key_value_heads": 8,
|
| 79 |
+
"rms_norm_eps": 1e-06,
|
| 80 |
+
"rope_scaling": null,
|
| 81 |
+
"rope_theta": 1000000,
|
| 82 |
+
"sliding_window": null,
|
| 83 |
+
"tie_word_embeddings": true,
|
| 84 |
+
"use_cache": true,
|
| 85 |
+
"use_sliding_window": false,
|
| 86 |
+
"vocab_size": 151936
|
| 87 |
+
}
|
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev3/qwen3_moe/optimum-internal-testing/tiny-random-qwen3_moe/b5678f2b1f926f36a4fd.json
ADDED
|
@@ -0,0 +1,65 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "optimum-internal-testing/tiny-random-qwen3_moe",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen3MoeForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"decoder_sparse_step": 2,
|
| 11 |
+
"dtype": "float32",
|
| 12 |
+
"head_dim": 32,
|
| 13 |
+
"hidden_act": "silu",
|
| 14 |
+
"hidden_size": 64,
|
| 15 |
+
"initializer_range": 0.02,
|
| 16 |
+
"intermediate_size": 128,
|
| 17 |
+
"max_position_embeddings": 40960,
|
| 18 |
+
"max_window_layers": 1,
|
| 19 |
+
"mlp_only_layers": [],
|
| 20 |
+
"model_type": "qwen3_moe",
|
| 21 |
+
"moe_intermediate_size": 128,
|
| 22 |
+
"neuron": {
|
| 23 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 24 |
+
"batch_size": 1,
|
| 25 |
+
"capacity_factor": null,
|
| 26 |
+
"checkpoint_id": "optimum-internal-testing/tiny-random-qwen3_moe",
|
| 27 |
+
"checkpoint_revision": "e0230be2839556b44b7400a233c73c74b4abb7af",
|
| 28 |
+
"continuous_batching": false,
|
| 29 |
+
"ep_degree": 1,
|
| 30 |
+
"fused_qkv": false,
|
| 31 |
+
"glu_mlp": true,
|
| 32 |
+
"local_ranks_size": 2,
|
| 33 |
+
"max_batch_size": 1,
|
| 34 |
+
"max_context_length": 1024,
|
| 35 |
+
"max_topk": 256,
|
| 36 |
+
"n_active_tokens": 1024,
|
| 37 |
+
"neuronxcc_version": "2.21.33363.0+82129205",
|
| 38 |
+
"on_device_sampling": true,
|
| 39 |
+
"optimum_neuron_version": "0.4.6.dev3",
|
| 40 |
+
"output_logits": false,
|
| 41 |
+
"pp_degree": 1,
|
| 42 |
+
"sequence_length": 1024,
|
| 43 |
+
"speculation_length": 0,
|
| 44 |
+
"start_rank_id": 0,
|
| 45 |
+
"target": "trn1",
|
| 46 |
+
"torch_dtype": "float32",
|
| 47 |
+
"tp_degree": 2
|
| 48 |
+
},
|
| 49 |
+
"norm_topk_prob": true,
|
| 50 |
+
"num_attention_heads": 2,
|
| 51 |
+
"num_experts": 8,
|
| 52 |
+
"num_experts_per_tok": 2,
|
| 53 |
+
"num_hidden_layers": 2,
|
| 54 |
+
"num_key_value_heads": 1,
|
| 55 |
+
"output_router_logits": false,
|
| 56 |
+
"rms_norm_eps": 1e-06,
|
| 57 |
+
"rope_scaling": null,
|
| 58 |
+
"rope_theta": 1000000.0,
|
| 59 |
+
"router_aux_loss_coef": 0.001,
|
| 60 |
+
"sliding_window": null,
|
| 61 |
+
"tie_word_embeddings": true,
|
| 62 |
+
"use_cache": true,
|
| 63 |
+
"use_sliding_window": false,
|
| 64 |
+
"vocab_size": 151936
|
| 65 |
+
}
|