Synchronizing local compiler cache.
Browse filesThis view is limited to 50 files because it contains too many changes.
See raw diff
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/granite/hf-internal-testing/tiny-random-GraniteForCausalLM/03ee1793b53b67fdba37.json +51 -0
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/granite/hf-internal-testing/tiny-random-GraniteForCausalLM/efcc0026ed5c762dea58.json +51 -0
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/granite/hf-internal-testing/tiny-random-GraniteForCausalLM/fc73a2065c050db68b09.json +51 -0
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/llama/llamafactory/tiny-random-Llama-3/5065586e7517bebdaf8e.json +55 -0
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/llama/llamafactory/tiny-random-Llama-3/57d9938920b5709ceeeb.json +55 -0
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/llama/llamafactory/tiny-random-Llama-3/6c7bb9378758475ee794.json +77 -0
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/llama/llamafactory/tiny-random-Llama-3/81866c45c8d5288cbdef.json +77 -0
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/llama/llamafactory/tiny-random-Llama-3/929f6cc6d81df5648af2.json +77 -0
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/llama/llamafactory/tiny-random-Llama-3/e76a1f3ff1d37decab49.json +55 -0
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/llama/unsloth/Llama-3.2-1B-Instruct/37ecc17a2362e0ea3fb1.json +78 -0
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/mixtral/dacorvo/Mixtral-tiny/48671ae7681a294bff2f.json +73 -0
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/mixtral/dacorvo/Mixtral-tiny/90d07ea993a4106ac65a.json +73 -0
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/mixtral/dacorvo/Mixtral-tiny/c8f95b53a594472481fe.json +73 -0
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/phi3/yujiepan/phi-4-tiny-random/1296bbdb99146d4b02ab.json +52 -0
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/phi3/yujiepan/phi-4-tiny-random/3ff6ca2a20ded0c5571c.json +52 -0
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/phi3/yujiepan/phi-4-tiny-random/b58cade3d8d5aaed2b99.json +52 -0
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/qwen2/yujiepan/qwen2.5-128k-tiny-random/6c77e99a6d0d386f0f92.json +53 -0
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/qwen2/yujiepan/qwen2.5-128k-tiny-random/896327ccd70bcb0e1606.json +53 -0
- neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/qwen2/yujiepan/qwen2.5-128k-tiny-random/984a30e1f373fb27c61e.json +53 -0
- neuronxcc-2.18.121.0+9e31e41a/MODULE_152c4fbd51eb1f6f90ec+84f3e719/model.hlo_module.pb +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_152c4fbd51eb1f6f90ec+84f3e719/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_1649fc77b87fff02e370+613edded/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_18642e0fd797db5b7fcb+431f5505/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_1b80b788e3a49498f963+613edded/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_1df250ef1cf7a7de560f+613edded/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_21d49e164d26352245e5+84f3e719/model.hlo_module.pb +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_21d49e164d26352245e5+84f3e719/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_22cf23062ec53b3fd95d+613edded/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_24ff9ac2787ce9a1d276+613edded/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_284ddd1b388e504631b8+ee23c5ad/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_284ddd1b388e504631b8+ee23c5ad/wrapped_neff.hlo +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_2ef52130792b59d66c66+613edded/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_320f2622d4d0c9fdd0f1+613edded/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_331276a07386ee77d52e+431f5505/model.neff +0 -0
- neuronxcc-2.18.121.0+9e31e41a/MODULE_34a6b42796c8b4e2f58b+431f5505/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_3cd14d7a79a82df7bd50+613edded/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_3da832fdaa3d62981800+613edded/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_40a0e75a65ac51fdd01a+613edded/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_48bfe9ceb9631fdca2d4+613edded/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_51d9fed86504dfbff43c+613edded/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_5368928916623911b1f2+84f3e719/model.hlo_module.pb +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_5368928916623911b1f2+84f3e719/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_54cb17f251d5b78afb71+6e42245d/model.hlo_module.pb +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_54cb17f251d5b78afb71+6e42245d/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_5c17a6fec29c60d2f8a5+6e42245d/model.hlo_module.pb +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_5c17a6fec29c60d2f8a5+6e42245d/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_73a8fcccb40e156a3330+6e42245d/model.hlo_module.pb +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_73a8fcccb40e156a3330+6e42245d/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_788eb0c6a9b0ca759eca+5be477de/model.neff +1 -1
- neuronxcc-2.18.121.0+9e31e41a/MODULE_788eb0c6a9b0ca759eca+5be477de/wrapped_neff.hlo +1 -1
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/granite/hf-internal-testing/tiny-random-GraniteForCausalLM/03ee1793b53b67fdba37.json
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "hf-internal-testing/tiny-random-GraniteForCausalLM",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"GraniteForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"attention_multiplier": 1.0,
|
| 11 |
+
"embedding_multiplier": 1.0,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 32,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 64,
|
| 16 |
+
"logits_scaling": 1.0,
|
| 17 |
+
"max_position_embeddings": 2048,
|
| 18 |
+
"mlp_bias": false,
|
| 19 |
+
"model_type": "granite",
|
| 20 |
+
"neuron": {
|
| 21 |
+
"_serialized_key": "HloNeuronConfig",
|
| 22 |
+
"all_reduce_dtype": null,
|
| 23 |
+
"allow_flash_attention": true,
|
| 24 |
+
"attention_layout": "HSB",
|
| 25 |
+
"attn_output_transposed": false,
|
| 26 |
+
"auto_cast_type": "bf16",
|
| 27 |
+
"batch_size": 1,
|
| 28 |
+
"checkpoint_id": "hf-internal-testing/tiny-random-GraniteForCausalLM",
|
| 29 |
+
"checkpoint_revision": "c3074ebc0ac2fe545305f5e5f6cce2cc9b2aa0c5",
|
| 30 |
+
"collectives_layout": "HSB",
|
| 31 |
+
"continuous_batching": false,
|
| 32 |
+
"fuse_qkv": true,
|
| 33 |
+
"group_query_attention": null,
|
| 34 |
+
"log_softmax_scores": false,
|
| 35 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 36 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 37 |
+
"output_all_logits": false,
|
| 38 |
+
"sequence_length": 100,
|
| 39 |
+
"tp_degree": 2
|
| 40 |
+
},
|
| 41 |
+
"num_attention_heads": 4,
|
| 42 |
+
"num_hidden_layers": 2,
|
| 43 |
+
"num_key_value_heads": 4,
|
| 44 |
+
"residual_multiplier": 1.0,
|
| 45 |
+
"rms_norm_eps": 1e-06,
|
| 46 |
+
"rope_scaling": null,
|
| 47 |
+
"rope_theta": 10000.0,
|
| 48 |
+
"tie_word_embeddings": false,
|
| 49 |
+
"use_cache": true,
|
| 50 |
+
"vocab_size": 49152
|
| 51 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/granite/hf-internal-testing/tiny-random-GraniteForCausalLM/efcc0026ed5c762dea58.json
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "hf-internal-testing/tiny-random-GraniteForCausalLM",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"GraniteForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"attention_multiplier": 1.0,
|
| 11 |
+
"embedding_multiplier": 1.0,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 32,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 64,
|
| 16 |
+
"logits_scaling": 1.0,
|
| 17 |
+
"max_position_embeddings": 2048,
|
| 18 |
+
"mlp_bias": false,
|
| 19 |
+
"model_type": "granite",
|
| 20 |
+
"neuron": {
|
| 21 |
+
"_serialized_key": "HloNeuronConfig",
|
| 22 |
+
"all_reduce_dtype": null,
|
| 23 |
+
"allow_flash_attention": true,
|
| 24 |
+
"attention_layout": "HSB",
|
| 25 |
+
"attn_output_transposed": false,
|
| 26 |
+
"auto_cast_type": "fp16",
|
| 27 |
+
"batch_size": 2,
|
| 28 |
+
"checkpoint_id": "hf-internal-testing/tiny-random-GraniteForCausalLM",
|
| 29 |
+
"checkpoint_revision": "c3074ebc0ac2fe545305f5e5f6cce2cc9b2aa0c5",
|
| 30 |
+
"collectives_layout": "HSB",
|
| 31 |
+
"continuous_batching": true,
|
| 32 |
+
"fuse_qkv": true,
|
| 33 |
+
"group_query_attention": null,
|
| 34 |
+
"log_softmax_scores": false,
|
| 35 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 36 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 37 |
+
"output_all_logits": false,
|
| 38 |
+
"sequence_length": 100,
|
| 39 |
+
"tp_degree": 2
|
| 40 |
+
},
|
| 41 |
+
"num_attention_heads": 4,
|
| 42 |
+
"num_hidden_layers": 2,
|
| 43 |
+
"num_key_value_heads": 4,
|
| 44 |
+
"residual_multiplier": 1.0,
|
| 45 |
+
"rms_norm_eps": 1e-06,
|
| 46 |
+
"rope_scaling": null,
|
| 47 |
+
"rope_theta": 10000.0,
|
| 48 |
+
"tie_word_embeddings": false,
|
| 49 |
+
"use_cache": true,
|
| 50 |
+
"vocab_size": 49152
|
| 51 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/granite/hf-internal-testing/tiny-random-GraniteForCausalLM/fc73a2065c050db68b09.json
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "hf-internal-testing/tiny-random-GraniteForCausalLM",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"GraniteForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"attention_multiplier": 1.0,
|
| 11 |
+
"embedding_multiplier": 1.0,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 32,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 64,
|
| 16 |
+
"logits_scaling": 1.0,
|
| 17 |
+
"max_position_embeddings": 2048,
|
| 18 |
+
"mlp_bias": false,
|
| 19 |
+
"model_type": "granite",
|
| 20 |
+
"neuron": {
|
| 21 |
+
"_serialized_key": "HloNeuronConfig",
|
| 22 |
+
"all_reduce_dtype": null,
|
| 23 |
+
"allow_flash_attention": true,
|
| 24 |
+
"attention_layout": "HSB",
|
| 25 |
+
"attn_output_transposed": false,
|
| 26 |
+
"auto_cast_type": "fp16",
|
| 27 |
+
"batch_size": 1,
|
| 28 |
+
"checkpoint_id": "hf-internal-testing/tiny-random-GraniteForCausalLM",
|
| 29 |
+
"checkpoint_revision": "c3074ebc0ac2fe545305f5e5f6cce2cc9b2aa0c5",
|
| 30 |
+
"collectives_layout": "HSB",
|
| 31 |
+
"continuous_batching": false,
|
| 32 |
+
"fuse_qkv": true,
|
| 33 |
+
"group_query_attention": null,
|
| 34 |
+
"log_softmax_scores": false,
|
| 35 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 36 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 37 |
+
"output_all_logits": false,
|
| 38 |
+
"sequence_length": 100,
|
| 39 |
+
"tp_degree": 2
|
| 40 |
+
},
|
| 41 |
+
"num_attention_heads": 4,
|
| 42 |
+
"num_hidden_layers": 2,
|
| 43 |
+
"num_key_value_heads": 4,
|
| 44 |
+
"residual_multiplier": 1.0,
|
| 45 |
+
"rms_norm_eps": 1e-06,
|
| 46 |
+
"rope_scaling": null,
|
| 47 |
+
"rope_theta": 10000.0,
|
| 48 |
+
"tie_word_embeddings": false,
|
| 49 |
+
"use_cache": true,
|
| 50 |
+
"vocab_size": 49152
|
| 51 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/llama/llamafactory/tiny-random-Llama-3/5065586e7517bebdaf8e.json
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "llamafactory/tiny-random-Llama-3",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"LlamaForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"head_dim": 4,
|
| 11 |
+
"hidden_act": "silu",
|
| 12 |
+
"hidden_size": 16,
|
| 13 |
+
"initializer_range": 0.02,
|
| 14 |
+
"intermediate_size": 64,
|
| 15 |
+
"max_position_embeddings": 131072,
|
| 16 |
+
"mlp_bias": false,
|
| 17 |
+
"model_type": "llama",
|
| 18 |
+
"neuron": {
|
| 19 |
+
"_serialized_key": "HloNeuronConfig",
|
| 20 |
+
"all_reduce_dtype": null,
|
| 21 |
+
"allow_flash_attention": true,
|
| 22 |
+
"attention_layout": "BSH",
|
| 23 |
+
"attn_output_transposed": false,
|
| 24 |
+
"auto_cast_type": "fp16",
|
| 25 |
+
"batch_size": 2,
|
| 26 |
+
"checkpoint_id": "llamafactory/tiny-random-Llama-3",
|
| 27 |
+
"checkpoint_revision": "bf2a2e3bf199ad2ee96f02a3c00246c608db22a8",
|
| 28 |
+
"collectives_layout": "HSB",
|
| 29 |
+
"continuous_batching": true,
|
| 30 |
+
"fuse_qkv": true,
|
| 31 |
+
"group_query_attention": null,
|
| 32 |
+
"log_softmax_scores": false,
|
| 33 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 34 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 35 |
+
"output_all_logits": false,
|
| 36 |
+
"sequence_length": 100,
|
| 37 |
+
"tp_degree": 2
|
| 38 |
+
},
|
| 39 |
+
"num_attention_heads": 4,
|
| 40 |
+
"num_hidden_layers": 2,
|
| 41 |
+
"num_key_value_heads": 4,
|
| 42 |
+
"pretraining_tp": 1,
|
| 43 |
+
"rms_norm_eps": 1e-05,
|
| 44 |
+
"rope_scaling": {
|
| 45 |
+
"factor": 8.0,
|
| 46 |
+
"high_freq_factor": 4.0,
|
| 47 |
+
"low_freq_factor": 1.0,
|
| 48 |
+
"original_max_position_embeddings": 8192,
|
| 49 |
+
"rope_type": "llama3"
|
| 50 |
+
},
|
| 51 |
+
"rope_theta": 500000.0,
|
| 52 |
+
"tie_word_embeddings": false,
|
| 53 |
+
"use_cache": true,
|
| 54 |
+
"vocab_size": 128256
|
| 55 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/llama/llamafactory/tiny-random-Llama-3/57d9938920b5709ceeeb.json
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "llamafactory/tiny-random-Llama-3",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"LlamaForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"head_dim": 4,
|
| 11 |
+
"hidden_act": "silu",
|
| 12 |
+
"hidden_size": 16,
|
| 13 |
+
"initializer_range": 0.02,
|
| 14 |
+
"intermediate_size": 64,
|
| 15 |
+
"max_position_embeddings": 131072,
|
| 16 |
+
"mlp_bias": false,
|
| 17 |
+
"model_type": "llama",
|
| 18 |
+
"neuron": {
|
| 19 |
+
"_serialized_key": "HloNeuronConfig",
|
| 20 |
+
"all_reduce_dtype": null,
|
| 21 |
+
"allow_flash_attention": true,
|
| 22 |
+
"attention_layout": "BSH",
|
| 23 |
+
"attn_output_transposed": false,
|
| 24 |
+
"auto_cast_type": "fp16",
|
| 25 |
+
"batch_size": 1,
|
| 26 |
+
"checkpoint_id": "llamafactory/tiny-random-Llama-3",
|
| 27 |
+
"checkpoint_revision": "bf2a2e3bf199ad2ee96f02a3c00246c608db22a8",
|
| 28 |
+
"collectives_layout": "HSB",
|
| 29 |
+
"continuous_batching": false,
|
| 30 |
+
"fuse_qkv": true,
|
| 31 |
+
"group_query_attention": null,
|
| 32 |
+
"log_softmax_scores": false,
|
| 33 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 34 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 35 |
+
"output_all_logits": false,
|
| 36 |
+
"sequence_length": 100,
|
| 37 |
+
"tp_degree": 2
|
| 38 |
+
},
|
| 39 |
+
"num_attention_heads": 4,
|
| 40 |
+
"num_hidden_layers": 2,
|
| 41 |
+
"num_key_value_heads": 4,
|
| 42 |
+
"pretraining_tp": 1,
|
| 43 |
+
"rms_norm_eps": 1e-05,
|
| 44 |
+
"rope_scaling": {
|
| 45 |
+
"factor": 8.0,
|
| 46 |
+
"high_freq_factor": 4.0,
|
| 47 |
+
"low_freq_factor": 1.0,
|
| 48 |
+
"original_max_position_embeddings": 8192,
|
| 49 |
+
"rope_type": "llama3"
|
| 50 |
+
},
|
| 51 |
+
"rope_theta": 500000.0,
|
| 52 |
+
"tie_word_embeddings": false,
|
| 53 |
+
"use_cache": true,
|
| 54 |
+
"vocab_size": 128256
|
| 55 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/llama/llamafactory/tiny-random-Llama-3/6c7bb9378758475ee794.json
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "llamafactory/tiny-random-Llama-3",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"LlamaForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"head_dim": 4,
|
| 11 |
+
"hidden_act": "silu",
|
| 12 |
+
"hidden_size": 16,
|
| 13 |
+
"initializer_range": 0.02,
|
| 14 |
+
"intermediate_size": 64,
|
| 15 |
+
"max_position_embeddings": 131072,
|
| 16 |
+
"mlp_bias": false,
|
| 17 |
+
"model_type": "llama",
|
| 18 |
+
"neuron": {
|
| 19 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 20 |
+
"async_mode": false,
|
| 21 |
+
"attn_kernel_enabled": false,
|
| 22 |
+
"batch_size": 1,
|
| 23 |
+
"capacity_factor": null,
|
| 24 |
+
"cc_pipeline_tiling_factor": 2,
|
| 25 |
+
"checkpoint_id": "llamafactory/tiny-random-Llama-3",
|
| 26 |
+
"checkpoint_revision": "bf2a2e3bf199ad2ee96f02a3c00246c608db22a8",
|
| 27 |
+
"continuous_batching": false,
|
| 28 |
+
"enable_bucketing": false,
|
| 29 |
+
"ep_degree": 1,
|
| 30 |
+
"flash_decoding_enabled": false,
|
| 31 |
+
"fused_qkv": true,
|
| 32 |
+
"glu_mlp": true,
|
| 33 |
+
"is_chunked_prefill": false,
|
| 34 |
+
"local_ranks_size": 2,
|
| 35 |
+
"logical_nc_config": 1,
|
| 36 |
+
"max_batch_size": 1,
|
| 37 |
+
"max_context_length": 100,
|
| 38 |
+
"max_topk": 256,
|
| 39 |
+
"mlp_kernel_enabled": false,
|
| 40 |
+
"mlp_kernel_fuse_residual_add": false,
|
| 41 |
+
"n_active_tokens": 100,
|
| 42 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 43 |
+
"num_cores_per_group": 1,
|
| 44 |
+
"on_device_sampling": true,
|
| 45 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 46 |
+
"output_logits": false,
|
| 47 |
+
"padding_side": "right",
|
| 48 |
+
"pp_degree": 1,
|
| 49 |
+
"qk_layernorm": false,
|
| 50 |
+
"qkv_kernel_enabled": false,
|
| 51 |
+
"rpl_reduce_dtype": "bfloat16",
|
| 52 |
+
"sequence_length": 100,
|
| 53 |
+
"sequence_parallel_enabled": false,
|
| 54 |
+
"speculation_length": 0,
|
| 55 |
+
"start_rank_id": 0,
|
| 56 |
+
"target": null,
|
| 57 |
+
"torch_dtype": "bfloat16",
|
| 58 |
+
"tp_degree": 2,
|
| 59 |
+
"vocab_parallel": false
|
| 60 |
+
},
|
| 61 |
+
"num_attention_heads": 4,
|
| 62 |
+
"num_hidden_layers": 2,
|
| 63 |
+
"num_key_value_heads": 4,
|
| 64 |
+
"pretraining_tp": 1,
|
| 65 |
+
"rms_norm_eps": 1e-05,
|
| 66 |
+
"rope_scaling": {
|
| 67 |
+
"factor": 8.0,
|
| 68 |
+
"high_freq_factor": 4.0,
|
| 69 |
+
"low_freq_factor": 1.0,
|
| 70 |
+
"original_max_position_embeddings": 8192,
|
| 71 |
+
"rope_type": "llama3"
|
| 72 |
+
},
|
| 73 |
+
"rope_theta": 500000.0,
|
| 74 |
+
"tie_word_embeddings": false,
|
| 75 |
+
"use_cache": true,
|
| 76 |
+
"vocab_size": 128256
|
| 77 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/llama/llamafactory/tiny-random-Llama-3/81866c45c8d5288cbdef.json
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "llamafactory/tiny-random-Llama-3",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"LlamaForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"head_dim": 4,
|
| 11 |
+
"hidden_act": "silu",
|
| 12 |
+
"hidden_size": 16,
|
| 13 |
+
"initializer_range": 0.02,
|
| 14 |
+
"intermediate_size": 64,
|
| 15 |
+
"max_position_embeddings": 131072,
|
| 16 |
+
"mlp_bias": false,
|
| 17 |
+
"model_type": "llama",
|
| 18 |
+
"neuron": {
|
| 19 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 20 |
+
"async_mode": false,
|
| 21 |
+
"attn_kernel_enabled": false,
|
| 22 |
+
"batch_size": 1,
|
| 23 |
+
"capacity_factor": null,
|
| 24 |
+
"cc_pipeline_tiling_factor": 2,
|
| 25 |
+
"checkpoint_id": "llamafactory/tiny-random-Llama-3",
|
| 26 |
+
"checkpoint_revision": "bf2a2e3bf199ad2ee96f02a3c00246c608db22a8",
|
| 27 |
+
"continuous_batching": false,
|
| 28 |
+
"enable_bucketing": false,
|
| 29 |
+
"ep_degree": 1,
|
| 30 |
+
"flash_decoding_enabled": false,
|
| 31 |
+
"fused_qkv": true,
|
| 32 |
+
"glu_mlp": true,
|
| 33 |
+
"is_chunked_prefill": false,
|
| 34 |
+
"local_ranks_size": 2,
|
| 35 |
+
"logical_nc_config": 1,
|
| 36 |
+
"max_batch_size": 1,
|
| 37 |
+
"max_context_length": 100,
|
| 38 |
+
"max_topk": 256,
|
| 39 |
+
"mlp_kernel_enabled": false,
|
| 40 |
+
"mlp_kernel_fuse_residual_add": false,
|
| 41 |
+
"n_active_tokens": 100,
|
| 42 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 43 |
+
"num_cores_per_group": 1,
|
| 44 |
+
"on_device_sampling": true,
|
| 45 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 46 |
+
"output_logits": false,
|
| 47 |
+
"padding_side": "right",
|
| 48 |
+
"pp_degree": 1,
|
| 49 |
+
"qk_layernorm": false,
|
| 50 |
+
"qkv_kernel_enabled": false,
|
| 51 |
+
"rpl_reduce_dtype": "float16",
|
| 52 |
+
"sequence_length": 100,
|
| 53 |
+
"sequence_parallel_enabled": false,
|
| 54 |
+
"speculation_length": 0,
|
| 55 |
+
"start_rank_id": 0,
|
| 56 |
+
"target": null,
|
| 57 |
+
"torch_dtype": "float16",
|
| 58 |
+
"tp_degree": 2,
|
| 59 |
+
"vocab_parallel": false
|
| 60 |
+
},
|
| 61 |
+
"num_attention_heads": 4,
|
| 62 |
+
"num_hidden_layers": 2,
|
| 63 |
+
"num_key_value_heads": 4,
|
| 64 |
+
"pretraining_tp": 1,
|
| 65 |
+
"rms_norm_eps": 1e-05,
|
| 66 |
+
"rope_scaling": {
|
| 67 |
+
"factor": 8.0,
|
| 68 |
+
"high_freq_factor": 4.0,
|
| 69 |
+
"low_freq_factor": 1.0,
|
| 70 |
+
"original_max_position_embeddings": 8192,
|
| 71 |
+
"rope_type": "llama3"
|
| 72 |
+
},
|
| 73 |
+
"rope_theta": 500000.0,
|
| 74 |
+
"tie_word_embeddings": false,
|
| 75 |
+
"use_cache": true,
|
| 76 |
+
"vocab_size": 128256
|
| 77 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/llama/llamafactory/tiny-random-Llama-3/929f6cc6d81df5648af2.json
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "llamafactory/tiny-random-Llama-3",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"LlamaForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"head_dim": 4,
|
| 11 |
+
"hidden_act": "silu",
|
| 12 |
+
"hidden_size": 16,
|
| 13 |
+
"initializer_range": 0.02,
|
| 14 |
+
"intermediate_size": 64,
|
| 15 |
+
"max_position_embeddings": 131072,
|
| 16 |
+
"mlp_bias": false,
|
| 17 |
+
"model_type": "llama",
|
| 18 |
+
"neuron": {
|
| 19 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 20 |
+
"async_mode": false,
|
| 21 |
+
"attn_kernel_enabled": false,
|
| 22 |
+
"batch_size": 2,
|
| 23 |
+
"capacity_factor": null,
|
| 24 |
+
"cc_pipeline_tiling_factor": 2,
|
| 25 |
+
"checkpoint_id": "llamafactory/tiny-random-Llama-3",
|
| 26 |
+
"checkpoint_revision": "bf2a2e3bf199ad2ee96f02a3c00246c608db22a8",
|
| 27 |
+
"continuous_batching": true,
|
| 28 |
+
"enable_bucketing": false,
|
| 29 |
+
"ep_degree": 1,
|
| 30 |
+
"flash_decoding_enabled": false,
|
| 31 |
+
"fused_qkv": true,
|
| 32 |
+
"glu_mlp": true,
|
| 33 |
+
"is_chunked_prefill": false,
|
| 34 |
+
"local_ranks_size": 2,
|
| 35 |
+
"logical_nc_config": 1,
|
| 36 |
+
"max_batch_size": 2,
|
| 37 |
+
"max_context_length": 100,
|
| 38 |
+
"max_topk": 256,
|
| 39 |
+
"mlp_kernel_enabled": false,
|
| 40 |
+
"mlp_kernel_fuse_residual_add": false,
|
| 41 |
+
"n_active_tokens": 100,
|
| 42 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 43 |
+
"num_cores_per_group": 1,
|
| 44 |
+
"on_device_sampling": false,
|
| 45 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 46 |
+
"output_logits": false,
|
| 47 |
+
"padding_side": "right",
|
| 48 |
+
"pp_degree": 1,
|
| 49 |
+
"qk_layernorm": false,
|
| 50 |
+
"qkv_kernel_enabled": false,
|
| 51 |
+
"rpl_reduce_dtype": "float16",
|
| 52 |
+
"sequence_length": 100,
|
| 53 |
+
"sequence_parallel_enabled": false,
|
| 54 |
+
"speculation_length": 0,
|
| 55 |
+
"start_rank_id": 0,
|
| 56 |
+
"target": null,
|
| 57 |
+
"torch_dtype": "float16",
|
| 58 |
+
"tp_degree": 2,
|
| 59 |
+
"vocab_parallel": false
|
| 60 |
+
},
|
| 61 |
+
"num_attention_heads": 4,
|
| 62 |
+
"num_hidden_layers": 2,
|
| 63 |
+
"num_key_value_heads": 4,
|
| 64 |
+
"pretraining_tp": 1,
|
| 65 |
+
"rms_norm_eps": 1e-05,
|
| 66 |
+
"rope_scaling": {
|
| 67 |
+
"factor": 8.0,
|
| 68 |
+
"high_freq_factor": 4.0,
|
| 69 |
+
"low_freq_factor": 1.0,
|
| 70 |
+
"original_max_position_embeddings": 8192,
|
| 71 |
+
"rope_type": "llama3"
|
| 72 |
+
},
|
| 73 |
+
"rope_theta": 500000.0,
|
| 74 |
+
"tie_word_embeddings": false,
|
| 75 |
+
"use_cache": true,
|
| 76 |
+
"vocab_size": 128256
|
| 77 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/llama/llamafactory/tiny-random-Llama-3/e76a1f3ff1d37decab49.json
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "llamafactory/tiny-random-Llama-3",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"LlamaForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"head_dim": 4,
|
| 11 |
+
"hidden_act": "silu",
|
| 12 |
+
"hidden_size": 16,
|
| 13 |
+
"initializer_range": 0.02,
|
| 14 |
+
"intermediate_size": 64,
|
| 15 |
+
"max_position_embeddings": 131072,
|
| 16 |
+
"mlp_bias": false,
|
| 17 |
+
"model_type": "llama",
|
| 18 |
+
"neuron": {
|
| 19 |
+
"_serialized_key": "HloNeuronConfig",
|
| 20 |
+
"all_reduce_dtype": null,
|
| 21 |
+
"allow_flash_attention": true,
|
| 22 |
+
"attention_layout": "BSH",
|
| 23 |
+
"attn_output_transposed": false,
|
| 24 |
+
"auto_cast_type": "bf16",
|
| 25 |
+
"batch_size": 1,
|
| 26 |
+
"checkpoint_id": "llamafactory/tiny-random-Llama-3",
|
| 27 |
+
"checkpoint_revision": "bf2a2e3bf199ad2ee96f02a3c00246c608db22a8",
|
| 28 |
+
"collectives_layout": "HSB",
|
| 29 |
+
"continuous_batching": false,
|
| 30 |
+
"fuse_qkv": true,
|
| 31 |
+
"group_query_attention": null,
|
| 32 |
+
"log_softmax_scores": false,
|
| 33 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 34 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 35 |
+
"output_all_logits": false,
|
| 36 |
+
"sequence_length": 100,
|
| 37 |
+
"tp_degree": 2
|
| 38 |
+
},
|
| 39 |
+
"num_attention_heads": 4,
|
| 40 |
+
"num_hidden_layers": 2,
|
| 41 |
+
"num_key_value_heads": 4,
|
| 42 |
+
"pretraining_tp": 1,
|
| 43 |
+
"rms_norm_eps": 1e-05,
|
| 44 |
+
"rope_scaling": {
|
| 45 |
+
"factor": 8.0,
|
| 46 |
+
"high_freq_factor": 4.0,
|
| 47 |
+
"low_freq_factor": 1.0,
|
| 48 |
+
"original_max_position_embeddings": 8192,
|
| 49 |
+
"rope_type": "llama3"
|
| 50 |
+
},
|
| 51 |
+
"rope_theta": 500000.0,
|
| 52 |
+
"tie_word_embeddings": false,
|
| 53 |
+
"use_cache": true,
|
| 54 |
+
"vocab_size": 128256
|
| 55 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/llama/unsloth/Llama-3.2-1B-Instruct/37ecc17a2362e0ea3fb1.json
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "unsloth/Llama-3.2-1B-Instruct",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"LlamaForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"head_dim": 64,
|
| 11 |
+
"hidden_act": "silu",
|
| 12 |
+
"hidden_size": 2048,
|
| 13 |
+
"initializer_range": 0.02,
|
| 14 |
+
"intermediate_size": 8192,
|
| 15 |
+
"max_position_embeddings": 131072,
|
| 16 |
+
"mlp_bias": false,
|
| 17 |
+
"model_type": "llama",
|
| 18 |
+
"neuron": {
|
| 19 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 20 |
+
"async_mode": false,
|
| 21 |
+
"attn_kernel_enabled": false,
|
| 22 |
+
"batch_size": 4,
|
| 23 |
+
"capacity_factor": null,
|
| 24 |
+
"cc_pipeline_tiling_factor": 2,
|
| 25 |
+
"checkpoint_id": "unsloth/Llama-3.2-1B-Instruct",
|
| 26 |
+
"checkpoint_revision": "5a8abab4a5d6f164389b1079fb721cfab8d7126c",
|
| 27 |
+
"continuous_batching": true,
|
| 28 |
+
"enable_bucketing": false,
|
| 29 |
+
"ep_degree": 1,
|
| 30 |
+
"flash_decoding_enabled": false,
|
| 31 |
+
"fused_qkv": true,
|
| 32 |
+
"glu_mlp": true,
|
| 33 |
+
"is_chunked_prefill": false,
|
| 34 |
+
"local_ranks_size": 2,
|
| 35 |
+
"logical_nc_config": 1,
|
| 36 |
+
"max_batch_size": 4,
|
| 37 |
+
"max_context_length": 4096,
|
| 38 |
+
"max_topk": 256,
|
| 39 |
+
"mlp_kernel_enabled": false,
|
| 40 |
+
"mlp_kernel_fuse_residual_add": false,
|
| 41 |
+
"n_active_tokens": 4096,
|
| 42 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 43 |
+
"num_cores_per_group": 1,
|
| 44 |
+
"on_device_sampling": false,
|
| 45 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 46 |
+
"output_logits": false,
|
| 47 |
+
"padding_side": "right",
|
| 48 |
+
"pp_degree": 1,
|
| 49 |
+
"qk_layernorm": false,
|
| 50 |
+
"qkv_kernel_enabled": false,
|
| 51 |
+
"rpl_reduce_dtype": "float16",
|
| 52 |
+
"sequence_length": 4096,
|
| 53 |
+
"sequence_parallel_enabled": false,
|
| 54 |
+
"speculation_length": 0,
|
| 55 |
+
"start_rank_id": 0,
|
| 56 |
+
"target": null,
|
| 57 |
+
"torch_dtype": "float16",
|
| 58 |
+
"tp_degree": 2,
|
| 59 |
+
"vocab_parallel": false
|
| 60 |
+
},
|
| 61 |
+
"num_attention_heads": 32,
|
| 62 |
+
"num_hidden_layers": 16,
|
| 63 |
+
"num_key_value_heads": 8,
|
| 64 |
+
"pretraining_tp": 1,
|
| 65 |
+
"rms_norm_eps": 1e-05,
|
| 66 |
+
"rope_scaling": {
|
| 67 |
+
"factor": 32.0,
|
| 68 |
+
"high_freq_factor": 4.0,
|
| 69 |
+
"low_freq_factor": 1.0,
|
| 70 |
+
"original_max_position_embeddings": 8192,
|
| 71 |
+
"rope_type": "llama3"
|
| 72 |
+
},
|
| 73 |
+
"rope_theta": 500000.0,
|
| 74 |
+
"tie_word_embeddings": true,
|
| 75 |
+
"unsloth_fixed": true,
|
| 76 |
+
"use_cache": true,
|
| 77 |
+
"vocab_size": 128256
|
| 78 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/mixtral/dacorvo/Mixtral-tiny/48671ae7681a294bff2f.json
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "dacorvo/Mixtral-tiny",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"MixtralForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_dropout": 0.0,
|
| 9 |
+
"head_dim": 32,
|
| 10 |
+
"hidden_act": "silu",
|
| 11 |
+
"hidden_size": 1024,
|
| 12 |
+
"initializer_range": 0.02,
|
| 13 |
+
"intermediate_size": 3584,
|
| 14 |
+
"max_position_embeddings": 1024,
|
| 15 |
+
"model_type": "mixtral",
|
| 16 |
+
"neuron": {
|
| 17 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 18 |
+
"async_mode": false,
|
| 19 |
+
"attn_kernel_enabled": false,
|
| 20 |
+
"batch_size": 1,
|
| 21 |
+
"capacity_factor": null,
|
| 22 |
+
"cc_pipeline_tiling_factor": 2,
|
| 23 |
+
"checkpoint_id": "dacorvo/Mixtral-tiny",
|
| 24 |
+
"checkpoint_revision": "c557ba205ddff6ea911f4719e0d543d6c08356b6",
|
| 25 |
+
"continuous_batching": false,
|
| 26 |
+
"enable_bucketing": false,
|
| 27 |
+
"ep_degree": 1,
|
| 28 |
+
"flash_decoding_enabled": false,
|
| 29 |
+
"fused_qkv": false,
|
| 30 |
+
"glu_mlp": true,
|
| 31 |
+
"is_chunked_prefill": false,
|
| 32 |
+
"local_ranks_size": 2,
|
| 33 |
+
"logical_nc_config": 1,
|
| 34 |
+
"max_batch_size": 1,
|
| 35 |
+
"max_context_length": 100,
|
| 36 |
+
"max_topk": 256,
|
| 37 |
+
"mlp_kernel_enabled": false,
|
| 38 |
+
"mlp_kernel_fuse_residual_add": false,
|
| 39 |
+
"n_active_tokens": 100,
|
| 40 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 41 |
+
"num_cores_per_group": 1,
|
| 42 |
+
"on_device_sampling": false,
|
| 43 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 44 |
+
"output_logits": false,
|
| 45 |
+
"padding_side": "right",
|
| 46 |
+
"pp_degree": 1,
|
| 47 |
+
"qk_layernorm": false,
|
| 48 |
+
"qkv_kernel_enabled": false,
|
| 49 |
+
"rpl_reduce_dtype": "bfloat16",
|
| 50 |
+
"sequence_length": 100,
|
| 51 |
+
"sequence_parallel_enabled": false,
|
| 52 |
+
"speculation_length": 0,
|
| 53 |
+
"start_rank_id": 0,
|
| 54 |
+
"target": null,
|
| 55 |
+
"torch_dtype": "bfloat16",
|
| 56 |
+
"tp_degree": 2,
|
| 57 |
+
"vocab_parallel": false
|
| 58 |
+
},
|
| 59 |
+
"num_attention_heads": 32,
|
| 60 |
+
"num_experts_per_tok": 2,
|
| 61 |
+
"num_hidden_layers": 2,
|
| 62 |
+
"num_key_value_heads": 8,
|
| 63 |
+
"num_local_experts": 8,
|
| 64 |
+
"output_router_logits": false,
|
| 65 |
+
"rms_norm_eps": 1e-05,
|
| 66 |
+
"rope_theta": 10000.0,
|
| 67 |
+
"router_aux_loss_coef": 0.001,
|
| 68 |
+
"router_jitter_noise": 0.0,
|
| 69 |
+
"sliding_window": 4096,
|
| 70 |
+
"tie_word_embeddings": false,
|
| 71 |
+
"use_cache": true,
|
| 72 |
+
"vocab_size": 32000
|
| 73 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/mixtral/dacorvo/Mixtral-tiny/90d07ea993a4106ac65a.json
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "dacorvo/Mixtral-tiny",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"MixtralForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_dropout": 0.0,
|
| 9 |
+
"head_dim": 32,
|
| 10 |
+
"hidden_act": "silu",
|
| 11 |
+
"hidden_size": 1024,
|
| 12 |
+
"initializer_range": 0.02,
|
| 13 |
+
"intermediate_size": 3584,
|
| 14 |
+
"max_position_embeddings": 1024,
|
| 15 |
+
"model_type": "mixtral",
|
| 16 |
+
"neuron": {
|
| 17 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 18 |
+
"async_mode": false,
|
| 19 |
+
"attn_kernel_enabled": false,
|
| 20 |
+
"batch_size": 1,
|
| 21 |
+
"capacity_factor": null,
|
| 22 |
+
"cc_pipeline_tiling_factor": 2,
|
| 23 |
+
"checkpoint_id": "dacorvo/Mixtral-tiny",
|
| 24 |
+
"checkpoint_revision": "c557ba205ddff6ea911f4719e0d543d6c08356b6",
|
| 25 |
+
"continuous_batching": false,
|
| 26 |
+
"enable_bucketing": false,
|
| 27 |
+
"ep_degree": 1,
|
| 28 |
+
"flash_decoding_enabled": false,
|
| 29 |
+
"fused_qkv": false,
|
| 30 |
+
"glu_mlp": true,
|
| 31 |
+
"is_chunked_prefill": false,
|
| 32 |
+
"local_ranks_size": 2,
|
| 33 |
+
"logical_nc_config": 1,
|
| 34 |
+
"max_batch_size": 1,
|
| 35 |
+
"max_context_length": 100,
|
| 36 |
+
"max_topk": 256,
|
| 37 |
+
"mlp_kernel_enabled": false,
|
| 38 |
+
"mlp_kernel_fuse_residual_add": false,
|
| 39 |
+
"n_active_tokens": 100,
|
| 40 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 41 |
+
"num_cores_per_group": 1,
|
| 42 |
+
"on_device_sampling": false,
|
| 43 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 44 |
+
"output_logits": false,
|
| 45 |
+
"padding_side": "right",
|
| 46 |
+
"pp_degree": 1,
|
| 47 |
+
"qk_layernorm": false,
|
| 48 |
+
"qkv_kernel_enabled": false,
|
| 49 |
+
"rpl_reduce_dtype": "float16",
|
| 50 |
+
"sequence_length": 100,
|
| 51 |
+
"sequence_parallel_enabled": false,
|
| 52 |
+
"speculation_length": 0,
|
| 53 |
+
"start_rank_id": 0,
|
| 54 |
+
"target": null,
|
| 55 |
+
"torch_dtype": "float16",
|
| 56 |
+
"tp_degree": 2,
|
| 57 |
+
"vocab_parallel": false
|
| 58 |
+
},
|
| 59 |
+
"num_attention_heads": 32,
|
| 60 |
+
"num_experts_per_tok": 2,
|
| 61 |
+
"num_hidden_layers": 2,
|
| 62 |
+
"num_key_value_heads": 8,
|
| 63 |
+
"num_local_experts": 8,
|
| 64 |
+
"output_router_logits": false,
|
| 65 |
+
"rms_norm_eps": 1e-05,
|
| 66 |
+
"rope_theta": 10000.0,
|
| 67 |
+
"router_aux_loss_coef": 0.001,
|
| 68 |
+
"router_jitter_noise": 0.0,
|
| 69 |
+
"sliding_window": 4096,
|
| 70 |
+
"tie_word_embeddings": false,
|
| 71 |
+
"use_cache": true,
|
| 72 |
+
"vocab_size": 32000
|
| 73 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/mixtral/dacorvo/Mixtral-tiny/c8f95b53a594472481fe.json
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "dacorvo/Mixtral-tiny",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"MixtralForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_dropout": 0.0,
|
| 9 |
+
"head_dim": 32,
|
| 10 |
+
"hidden_act": "silu",
|
| 11 |
+
"hidden_size": 1024,
|
| 12 |
+
"initializer_range": 0.02,
|
| 13 |
+
"intermediate_size": 3584,
|
| 14 |
+
"max_position_embeddings": 1024,
|
| 15 |
+
"model_type": "mixtral",
|
| 16 |
+
"neuron": {
|
| 17 |
+
"_serialized_key": "NxDNeuronConfig",
|
| 18 |
+
"async_mode": false,
|
| 19 |
+
"attn_kernel_enabled": false,
|
| 20 |
+
"batch_size": 2,
|
| 21 |
+
"capacity_factor": null,
|
| 22 |
+
"cc_pipeline_tiling_factor": 2,
|
| 23 |
+
"checkpoint_id": "dacorvo/Mixtral-tiny",
|
| 24 |
+
"checkpoint_revision": "c557ba205ddff6ea911f4719e0d543d6c08356b6",
|
| 25 |
+
"continuous_batching": false,
|
| 26 |
+
"enable_bucketing": false,
|
| 27 |
+
"ep_degree": 1,
|
| 28 |
+
"flash_decoding_enabled": false,
|
| 29 |
+
"fused_qkv": false,
|
| 30 |
+
"glu_mlp": true,
|
| 31 |
+
"is_chunked_prefill": false,
|
| 32 |
+
"local_ranks_size": 2,
|
| 33 |
+
"logical_nc_config": 1,
|
| 34 |
+
"max_batch_size": 2,
|
| 35 |
+
"max_context_length": 100,
|
| 36 |
+
"max_topk": 256,
|
| 37 |
+
"mlp_kernel_enabled": false,
|
| 38 |
+
"mlp_kernel_fuse_residual_add": false,
|
| 39 |
+
"n_active_tokens": 100,
|
| 40 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 41 |
+
"num_cores_per_group": 1,
|
| 42 |
+
"on_device_sampling": false,
|
| 43 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 44 |
+
"output_logits": false,
|
| 45 |
+
"padding_side": "right",
|
| 46 |
+
"pp_degree": 1,
|
| 47 |
+
"qk_layernorm": false,
|
| 48 |
+
"qkv_kernel_enabled": false,
|
| 49 |
+
"rpl_reduce_dtype": "float16",
|
| 50 |
+
"sequence_length": 100,
|
| 51 |
+
"sequence_parallel_enabled": false,
|
| 52 |
+
"speculation_length": 0,
|
| 53 |
+
"start_rank_id": 0,
|
| 54 |
+
"target": null,
|
| 55 |
+
"torch_dtype": "float16",
|
| 56 |
+
"tp_degree": 2,
|
| 57 |
+
"vocab_parallel": false
|
| 58 |
+
},
|
| 59 |
+
"num_attention_heads": 32,
|
| 60 |
+
"num_experts_per_tok": 2,
|
| 61 |
+
"num_hidden_layers": 2,
|
| 62 |
+
"num_key_value_heads": 8,
|
| 63 |
+
"num_local_experts": 8,
|
| 64 |
+
"output_router_logits": false,
|
| 65 |
+
"rms_norm_eps": 1e-05,
|
| 66 |
+
"rope_theta": 10000.0,
|
| 67 |
+
"router_aux_loss_coef": 0.001,
|
| 68 |
+
"router_jitter_noise": 0.0,
|
| 69 |
+
"sliding_window": 4096,
|
| 70 |
+
"tie_word_embeddings": false,
|
| 71 |
+
"use_cache": true,
|
| 72 |
+
"vocab_size": 32000
|
| 73 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/phi3/yujiepan/phi-4-tiny-random/1296bbdb99146d4b02ab.json
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "yujiepan/phi-4-tiny-random",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Phi3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"auto_map": {},
|
| 11 |
+
"embd_pdrop": 0.0,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 16,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 32,
|
| 16 |
+
"max_position_embeddings": 16384,
|
| 17 |
+
"model_type": "phi3",
|
| 18 |
+
"neuron": {
|
| 19 |
+
"_serialized_key": "HloNeuronConfig",
|
| 20 |
+
"all_reduce_dtype": null,
|
| 21 |
+
"allow_flash_attention": false,
|
| 22 |
+
"attention_layout": "HSB",
|
| 23 |
+
"attn_output_transposed": false,
|
| 24 |
+
"auto_cast_type": "bf16",
|
| 25 |
+
"batch_size": 1,
|
| 26 |
+
"checkpoint_id": "yujiepan/phi-4-tiny-random",
|
| 27 |
+
"checkpoint_revision": "18a9a1168dc97ac6d128f811925670c275610f5a",
|
| 28 |
+
"collectives_layout": "HSB",
|
| 29 |
+
"continuous_batching": false,
|
| 30 |
+
"fuse_qkv": true,
|
| 31 |
+
"group_query_attention": "replicated-heads",
|
| 32 |
+
"log_softmax_scores": false,
|
| 33 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 34 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 35 |
+
"output_all_logits": false,
|
| 36 |
+
"sequence_length": 100,
|
| 37 |
+
"tp_degree": 2
|
| 38 |
+
},
|
| 39 |
+
"num_attention_heads": 2,
|
| 40 |
+
"num_hidden_layers": 2,
|
| 41 |
+
"num_key_value_heads": 1,
|
| 42 |
+
"original_max_position_embeddings": 16384,
|
| 43 |
+
"partial_rotary_factor": 1.0,
|
| 44 |
+
"resid_pdrop": 0.0,
|
| 45 |
+
"rms_norm_eps": 1e-05,
|
| 46 |
+
"rope_scaling": null,
|
| 47 |
+
"rope_theta": 250000,
|
| 48 |
+
"sliding_window": null,
|
| 49 |
+
"tie_word_embeddings": false,
|
| 50 |
+
"use_cache": true,
|
| 51 |
+
"vocab_size": 100352
|
| 52 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/phi3/yujiepan/phi-4-tiny-random/3ff6ca2a20ded0c5571c.json
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "yujiepan/phi-4-tiny-random",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Phi3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"auto_map": {},
|
| 11 |
+
"embd_pdrop": 0.0,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 16,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 32,
|
| 16 |
+
"max_position_embeddings": 16384,
|
| 17 |
+
"model_type": "phi3",
|
| 18 |
+
"neuron": {
|
| 19 |
+
"_serialized_key": "HloNeuronConfig",
|
| 20 |
+
"all_reduce_dtype": null,
|
| 21 |
+
"allow_flash_attention": false,
|
| 22 |
+
"attention_layout": "HSB",
|
| 23 |
+
"attn_output_transposed": false,
|
| 24 |
+
"auto_cast_type": "fp16",
|
| 25 |
+
"batch_size": 2,
|
| 26 |
+
"checkpoint_id": "yujiepan/phi-4-tiny-random",
|
| 27 |
+
"checkpoint_revision": "18a9a1168dc97ac6d128f811925670c275610f5a",
|
| 28 |
+
"collectives_layout": "HSB",
|
| 29 |
+
"continuous_batching": true,
|
| 30 |
+
"fuse_qkv": true,
|
| 31 |
+
"group_query_attention": "replicated-heads",
|
| 32 |
+
"log_softmax_scores": false,
|
| 33 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 34 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 35 |
+
"output_all_logits": false,
|
| 36 |
+
"sequence_length": 100,
|
| 37 |
+
"tp_degree": 2
|
| 38 |
+
},
|
| 39 |
+
"num_attention_heads": 2,
|
| 40 |
+
"num_hidden_layers": 2,
|
| 41 |
+
"num_key_value_heads": 1,
|
| 42 |
+
"original_max_position_embeddings": 16384,
|
| 43 |
+
"partial_rotary_factor": 1.0,
|
| 44 |
+
"resid_pdrop": 0.0,
|
| 45 |
+
"rms_norm_eps": 1e-05,
|
| 46 |
+
"rope_scaling": null,
|
| 47 |
+
"rope_theta": 250000,
|
| 48 |
+
"sliding_window": null,
|
| 49 |
+
"tie_word_embeddings": false,
|
| 50 |
+
"use_cache": true,
|
| 51 |
+
"vocab_size": 100352
|
| 52 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/phi3/yujiepan/phi-4-tiny-random/b58cade3d8d5aaed2b99.json
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "yujiepan/phi-4-tiny-random",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Phi3ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_bias": false,
|
| 9 |
+
"attention_dropout": 0.0,
|
| 10 |
+
"auto_map": {},
|
| 11 |
+
"embd_pdrop": 0.0,
|
| 12 |
+
"hidden_act": "silu",
|
| 13 |
+
"hidden_size": 16,
|
| 14 |
+
"initializer_range": 0.02,
|
| 15 |
+
"intermediate_size": 32,
|
| 16 |
+
"max_position_embeddings": 16384,
|
| 17 |
+
"model_type": "phi3",
|
| 18 |
+
"neuron": {
|
| 19 |
+
"_serialized_key": "HloNeuronConfig",
|
| 20 |
+
"all_reduce_dtype": null,
|
| 21 |
+
"allow_flash_attention": false,
|
| 22 |
+
"attention_layout": "HSB",
|
| 23 |
+
"attn_output_transposed": false,
|
| 24 |
+
"auto_cast_type": "fp16",
|
| 25 |
+
"batch_size": 1,
|
| 26 |
+
"checkpoint_id": "yujiepan/phi-4-tiny-random",
|
| 27 |
+
"checkpoint_revision": "18a9a1168dc97ac6d128f811925670c275610f5a",
|
| 28 |
+
"collectives_layout": "HSB",
|
| 29 |
+
"continuous_batching": false,
|
| 30 |
+
"fuse_qkv": true,
|
| 31 |
+
"group_query_attention": "replicated-heads",
|
| 32 |
+
"log_softmax_scores": false,
|
| 33 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 34 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 35 |
+
"output_all_logits": false,
|
| 36 |
+
"sequence_length": 100,
|
| 37 |
+
"tp_degree": 2
|
| 38 |
+
},
|
| 39 |
+
"num_attention_heads": 2,
|
| 40 |
+
"num_hidden_layers": 2,
|
| 41 |
+
"num_key_value_heads": 1,
|
| 42 |
+
"original_max_position_embeddings": 16384,
|
| 43 |
+
"partial_rotary_factor": 1.0,
|
| 44 |
+
"resid_pdrop": 0.0,
|
| 45 |
+
"rms_norm_eps": 1e-05,
|
| 46 |
+
"rope_scaling": null,
|
| 47 |
+
"rope_theta": 250000,
|
| 48 |
+
"sliding_window": null,
|
| 49 |
+
"tie_word_embeddings": false,
|
| 50 |
+
"use_cache": true,
|
| 51 |
+
"vocab_size": 100352
|
| 52 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/qwen2/yujiepan/qwen2.5-128k-tiny-random/6c77e99a6d0d386f0f92.json
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "yujiepan/qwen2.5-128k-tiny-random",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen2ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_dropout": 0.0,
|
| 9 |
+
"hidden_act": "silu",
|
| 10 |
+
"hidden_size": 8,
|
| 11 |
+
"initializer_range": 0.02,
|
| 12 |
+
"intermediate_size": 16,
|
| 13 |
+
"max_position_embeddings": 32768,
|
| 14 |
+
"max_window_layers": 1,
|
| 15 |
+
"model_type": "qwen2",
|
| 16 |
+
"neuron": {
|
| 17 |
+
"_serialized_key": "HloNeuronConfig",
|
| 18 |
+
"all_reduce_dtype": null,
|
| 19 |
+
"allow_flash_attention": true,
|
| 20 |
+
"attention_layout": "HSB",
|
| 21 |
+
"attn_output_transposed": false,
|
| 22 |
+
"auto_cast_type": "fp16",
|
| 23 |
+
"batch_size": 1,
|
| 24 |
+
"checkpoint_id": "yujiepan/qwen2.5-128k-tiny-random",
|
| 25 |
+
"checkpoint_revision": "c8296d4ca3f87782876d2382fbb6481d1beb8ef0",
|
| 26 |
+
"collectives_layout": "HSB",
|
| 27 |
+
"continuous_batching": false,
|
| 28 |
+
"fuse_qkv": false,
|
| 29 |
+
"group_query_attention": "shard-over-heads",
|
| 30 |
+
"log_softmax_scores": false,
|
| 31 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 32 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 33 |
+
"output_all_logits": false,
|
| 34 |
+
"sequence_length": 100,
|
| 35 |
+
"tp_degree": 2
|
| 36 |
+
},
|
| 37 |
+
"num_attention_heads": 4,
|
| 38 |
+
"num_hidden_layers": 2,
|
| 39 |
+
"num_key_value_heads": 2,
|
| 40 |
+
"rms_norm_eps": 1e-06,
|
| 41 |
+
"rope_scaling": {
|
| 42 |
+
"factor": 4.0,
|
| 43 |
+
"original_max_position_embeddings": 32768,
|
| 44 |
+
"rope_type": "yarn",
|
| 45 |
+
"type": "yarn"
|
| 46 |
+
},
|
| 47 |
+
"rope_theta": 1000000.0,
|
| 48 |
+
"sliding_window": 131072,
|
| 49 |
+
"tie_word_embeddings": false,
|
| 50 |
+
"use_cache": true,
|
| 51 |
+
"use_sliding_window": false,
|
| 52 |
+
"vocab_size": 152064
|
| 53 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/qwen2/yujiepan/qwen2.5-128k-tiny-random/896327ccd70bcb0e1606.json
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "yujiepan/qwen2.5-128k-tiny-random",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen2ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_dropout": 0.0,
|
| 9 |
+
"hidden_act": "silu",
|
| 10 |
+
"hidden_size": 8,
|
| 11 |
+
"initializer_range": 0.02,
|
| 12 |
+
"intermediate_size": 16,
|
| 13 |
+
"max_position_embeddings": 32768,
|
| 14 |
+
"max_window_layers": 1,
|
| 15 |
+
"model_type": "qwen2",
|
| 16 |
+
"neuron": {
|
| 17 |
+
"_serialized_key": "HloNeuronConfig",
|
| 18 |
+
"all_reduce_dtype": null,
|
| 19 |
+
"allow_flash_attention": true,
|
| 20 |
+
"attention_layout": "HSB",
|
| 21 |
+
"attn_output_transposed": false,
|
| 22 |
+
"auto_cast_type": "bf16",
|
| 23 |
+
"batch_size": 1,
|
| 24 |
+
"checkpoint_id": "yujiepan/qwen2.5-128k-tiny-random",
|
| 25 |
+
"checkpoint_revision": "c8296d4ca3f87782876d2382fbb6481d1beb8ef0",
|
| 26 |
+
"collectives_layout": "HSB",
|
| 27 |
+
"continuous_batching": false,
|
| 28 |
+
"fuse_qkv": false,
|
| 29 |
+
"group_query_attention": "shard-over-heads",
|
| 30 |
+
"log_softmax_scores": false,
|
| 31 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 32 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 33 |
+
"output_all_logits": false,
|
| 34 |
+
"sequence_length": 100,
|
| 35 |
+
"tp_degree": 2
|
| 36 |
+
},
|
| 37 |
+
"num_attention_heads": 4,
|
| 38 |
+
"num_hidden_layers": 2,
|
| 39 |
+
"num_key_value_heads": 2,
|
| 40 |
+
"rms_norm_eps": 1e-06,
|
| 41 |
+
"rope_scaling": {
|
| 42 |
+
"factor": 4.0,
|
| 43 |
+
"original_max_position_embeddings": 32768,
|
| 44 |
+
"rope_type": "yarn",
|
| 45 |
+
"type": "yarn"
|
| 46 |
+
},
|
| 47 |
+
"rope_theta": 1000000.0,
|
| 48 |
+
"sliding_window": 131072,
|
| 49 |
+
"tie_word_embeddings": false,
|
| 50 |
+
"use_cache": true,
|
| 51 |
+
"use_sliding_window": false,
|
| 52 |
+
"vocab_size": 152064
|
| 53 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/0_REGISTRY/0.3.0.dev1/qwen2/yujiepan/qwen2.5-128k-tiny-random/984a30e1f373fb27c61e.json
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_entry_class": "SingleModelCacheEntry",
|
| 3 |
+
"_model_id": "yujiepan/qwen2.5-128k-tiny-random",
|
| 4 |
+
"_task": "text-generation",
|
| 5 |
+
"architectures": [
|
| 6 |
+
"Qwen2ForCausalLM"
|
| 7 |
+
],
|
| 8 |
+
"attention_dropout": 0.0,
|
| 9 |
+
"hidden_act": "silu",
|
| 10 |
+
"hidden_size": 8,
|
| 11 |
+
"initializer_range": 0.02,
|
| 12 |
+
"intermediate_size": 16,
|
| 13 |
+
"max_position_embeddings": 32768,
|
| 14 |
+
"max_window_layers": 1,
|
| 15 |
+
"model_type": "qwen2",
|
| 16 |
+
"neuron": {
|
| 17 |
+
"_serialized_key": "HloNeuronConfig",
|
| 18 |
+
"all_reduce_dtype": null,
|
| 19 |
+
"allow_flash_attention": true,
|
| 20 |
+
"attention_layout": "HSB",
|
| 21 |
+
"attn_output_transposed": false,
|
| 22 |
+
"auto_cast_type": "fp16",
|
| 23 |
+
"batch_size": 2,
|
| 24 |
+
"checkpoint_id": "yujiepan/qwen2.5-128k-tiny-random",
|
| 25 |
+
"checkpoint_revision": "c8296d4ca3f87782876d2382fbb6481d1beb8ef0",
|
| 26 |
+
"collectives_layout": "HSB",
|
| 27 |
+
"continuous_batching": true,
|
| 28 |
+
"fuse_qkv": false,
|
| 29 |
+
"group_query_attention": "shard-over-heads",
|
| 30 |
+
"log_softmax_scores": false,
|
| 31 |
+
"neuronxcc_version": "2.18.121.0+9e31e41a",
|
| 32 |
+
"optimum_neuron_version": "0.3.0.dev1",
|
| 33 |
+
"output_all_logits": false,
|
| 34 |
+
"sequence_length": 100,
|
| 35 |
+
"tp_degree": 2
|
| 36 |
+
},
|
| 37 |
+
"num_attention_heads": 4,
|
| 38 |
+
"num_hidden_layers": 2,
|
| 39 |
+
"num_key_value_heads": 2,
|
| 40 |
+
"rms_norm_eps": 1e-06,
|
| 41 |
+
"rope_scaling": {
|
| 42 |
+
"factor": 4.0,
|
| 43 |
+
"original_max_position_embeddings": 32768,
|
| 44 |
+
"rope_type": "yarn",
|
| 45 |
+
"type": "yarn"
|
| 46 |
+
},
|
| 47 |
+
"rope_theta": 1000000.0,
|
| 48 |
+
"sliding_window": 131072,
|
| 49 |
+
"tie_word_embeddings": false,
|
| 50 |
+
"use_cache": true,
|
| 51 |
+
"use_sliding_window": false,
|
| 52 |
+
"vocab_size": 152064
|
| 53 |
+
}
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_152c4fbd51eb1f6f90ec+84f3e719/model.hlo_module.pb
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 80975
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ca6707e24b81612b57e4fb439ffd77ed7a8e7464b09f782a6fbd4dd087ad261f
|
| 3 |
size 80975
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_152c4fbd51eb1f6f90ec+84f3e719/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 226304
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:8207cd1aa055aea2f4e9bcf0a1b677720683d9b6e210f9a2c47c19fe07c741ec
|
| 3 |
size 226304
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_1649fc77b87fff02e370+613edded/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 134144
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:712cc0564a70d624be4585d9d0bf9e6da48ba2ab344aaa46b3bc2b69d0fb65b2
|
| 3 |
size 134144
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_18642e0fd797db5b7fcb+431f5505/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 103424
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6ea7a800ae23210b77f9ca5af753770e7ce3f20cac52a5cbfb6308b2b316afc0
|
| 3 |
size 103424
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_1b80b788e3a49498f963+613edded/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 144384
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:209ce50c4ff21f9cccf11aa6c3c9fe8b56a458b85f915ae726b9e6e618bc10e3
|
| 3 |
size 144384
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_1df250ef1cf7a7de560f+613edded/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 134144
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4a8e5a4b5ae9b7430b8f1d4af36247bdecb527e16b6e3522e7bf1f8d598bdce2
|
| 3 |
size 134144
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_21d49e164d26352245e5+84f3e719/model.hlo_module.pb
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 448722
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f3ccb7b89f7123461216316af02734622aacbdaaf771430310b55142ec4a8f66
|
| 3 |
size 448722
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_21d49e164d26352245e5+84f3e719/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 32257024
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f309ec2855b27f9f1c7e04499d51dc3a1acc033f86011ba5fa360d32b900575f
|
| 3 |
size 32257024
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_22cf23062ec53b3fd95d+613edded/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 144384
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:cdff629fd4875b4758c2e7689d06fbf4d3f6af10147990e2e59d23ae0306dc56
|
| 3 |
size 144384
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_24ff9ac2787ce9a1d276+613edded/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 134144
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:abf956fe48036d70745811da6e26b78a5cd0d05da19e7d450650ad9e91afb527
|
| 3 |
size 134144
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_284ddd1b388e504631b8+ee23c5ad/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 257024
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1d3b6d3a13fd959c5d03095ca331b2f52bb8e851678a6a6eff44dc1ac703dfed
|
| 3 |
size 257024
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_284ddd1b388e504631b8+ee23c5ad/wrapped_neff.hlo
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 268322
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:38efd1d0e95256e32d7c73e4e8b3c7bd98e9a56871c84699c620d5cd1920e304
|
| 3 |
size 268322
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_2ef52130792b59d66c66+613edded/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 134144
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c11f49ac265b6a8d5857e4bec676fc5ac5254e40aa32cc36531aabc2aee91214
|
| 3 |
size 134144
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_320f2622d4d0c9fdd0f1+613edded/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 144384
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:65de3b8fd9d7a5337eabaa7893537714a910ed599cb2ee67e59293f7a36da99e
|
| 3 |
size 144384
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_331276a07386ee77d52e+431f5505/model.neff
CHANGED
|
Binary files a/neuronxcc-2.18.121.0+9e31e41a/MODULE_331276a07386ee77d52e+431f5505/model.neff and b/neuronxcc-2.18.121.0+9e31e41a/MODULE_331276a07386ee77d52e+431f5505/model.neff differ
|
|
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_34a6b42796c8b4e2f58b+431f5505/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1158144
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b93fabc1801093a09caea0fa761a39b924a9319556993ae785e20c78241e0f9c
|
| 3 |
size 1158144
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_3cd14d7a79a82df7bd50+613edded/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 144384
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0d6aeb3f38b07b2fdcf6f2e680c0444d40989ae6046cbe947d6c8c1b38fcfa0a
|
| 3 |
size 144384
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_3da832fdaa3d62981800+613edded/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 154624
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4537046a6a43824c15281096e8092df328725b74b4d983a7ce4a20d9eabe530f
|
| 3 |
size 154624
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_40a0e75a65ac51fdd01a+613edded/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 144384
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:af9bda725f1fb52ebbc3ccdb025ca435b68543fa8c8db4dfc4dc7550f56dcb40
|
| 3 |
size 144384
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_48bfe9ceb9631fdca2d4+613edded/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 144384
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4c829e29b801ab1c55fa0203652e35021a92a3fb9764cfee430ea68a64ba1f56
|
| 3 |
size 144384
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_51d9fed86504dfbff43c+613edded/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 134144
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e53b8fb997dc2fcb8759f10a88c687593562cf6bf7bea98d9dd7858a0d508aa4
|
| 3 |
size 134144
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_5368928916623911b1f2+84f3e719/model.hlo_module.pb
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 47194
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2f60b69002d5b5cc2b0e7d9c2da141c97b7b52df0a75908b882ce4a26aa4de3d
|
| 3 |
size 47194
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_5368928916623911b1f2+84f3e719/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 246784
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a19a5a0a62270b6eab7786a803cdc5049f57c2bd5ffe8dbcb3e4f3206ca569de
|
| 3 |
size 246784
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_54cb17f251d5b78afb71+6e42245d/model.hlo_module.pb
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 81317
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:21cf5c027b34406011bdb0ac7ae8f0de44055c4bc5e4d2cecadbe543b63330e6
|
| 3 |
size 81317
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_54cb17f251d5b78afb71+6e42245d/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 359424
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f8609cb1efa5b4602fd283d7f2c3813a455452a9159ffa9b3616cdacb0919c42
|
| 3 |
size 359424
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_5c17a6fec29c60d2f8a5+6e42245d/model.hlo_module.pb
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 81319
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:14a0969992856b5db95988762f04fb2bc9bb54bc6255fb2d71cd6740b2db743a
|
| 3 |
size 81319
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_5c17a6fec29c60d2f8a5+6e42245d/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 359424
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:85bffa860c95571e7a070d4f956f9dd28b4b101676fc4316d73c4fe8be375606
|
| 3 |
size 359424
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_73a8fcccb40e156a3330+6e42245d/model.hlo_module.pb
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 82168
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e13cd93afd9d1e18b9ce7df8249f5d7d9b950369a162e113a1fc0bd765a22bb6
|
| 3 |
size 82168
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_73a8fcccb40e156a3330+6e42245d/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 420864
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2032a8dc8a210e031f0abfcff0c1b170a94bd40cf549c20de09ac6211d1f3888
|
| 3 |
size 420864
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_788eb0c6a9b0ca759eca+5be477de/model.neff
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 2386944
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:422fe9682a9a014ee444b66f45e6863666481346558b956507344d5b6e122b51
|
| 3 |
size 2386944
|
neuronxcc-2.18.121.0+9e31e41a/MODULE_788eb0c6a9b0ca759eca+5be477de/wrapped_neff.hlo
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 2394734
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:60bd55d6b3ed01569c8b1a01a8582321a55924ddcdac7f6da53906d5dde8d461
|
| 3 |
size 2394734
|