Synchronizing local compiler cache.

Files changed (1) hide show

neuronxcc-2.17.194.0+d312836f/0_REGISTRY/0.2.0.dev6/llama/TinyLlama/TinyLlama-1.1B-Chat-v1.0/9790d481e3a3e01abad0.json ADDED Viewed

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "TinyLlama/TinyLlama-1.1B-Chat-v1.0",
+  "_task": "text-generation",
+  "architectures": [
+    "LlamaForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "head_dim": 64,
+  "hidden_act": "silu",
+  "hidden_size": 2048,
+  "initializer_range": 0.02,
+  "intermediate_size": 5632,
+  "max_position_embeddings": 2048,
+  "mlp_bias": false,
+  "model_type": "llama",
+  "neuron": {
+    "_serialized_key": "HloNeuronConfig",
+    "all_reduce_dtype": null,
+    "allow_flash_attention": true,
+    "attention_layout": "BSH",
+    "attn_output_transposed": false,
+    "auto_cast_type": "bf16",
+    "batch_size": 1,
+    "checkpoint_id": "TinyLlama/TinyLlama-1.1B-Chat-v1.0",
+    "checkpoint_revision": "fe8a4ea1ffedaf415f4da2f062534de366a451e6",
+    "collectives_layout": "HSB",
+    "continuous_batching": false,
+    "fuse_qkv": true,
+    "group_query_attention": "shard-over-heads",
+    "log_softmax_scores": false,
+    "neuronxcc_version": "2.17.194.0+d312836f",
+    "optimum_neuron_version": "0.2.0.dev6",
+    "output_all_logits": false,
+    "sequence_length": 2048,
+    "tp_degree": 2
+  },
+  "num_attention_heads": 32,
+  "num_hidden_layers": 22,
+  "num_key_value_heads": 4,
+  "pretraining_tp": 1,
+  "rms_norm_eps": 1e-05,
+  "rope_scaling": null,
+  "rope_theta": 10000.0,
+  "tie_word_embeddings": false,
+  "use_cache": true,
+  "vocab_size": 32000
+}