dacorvo HF Staff commited on 10 days ago

Commit

700ac84

verified ·

1 Parent(s): ab5eea5

Synchronizing local compiler cache.

Browse files

Files changed (41) hide show

.gitattributes +5 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/03b0c107d1cede36875199a5d51decfe04c473de2af9999f8577a028d74d0ab4/2339cde11cbd17accc62.json +189 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/2da7a00f0478d50ae1e7f75f085c5b2773b5f355f427c61cf34cb6febd629d96/7d7f3f7850d2aa89e3fa.json +60 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/441269935591cad8d370e512c0b93cdd2fce6247c40e5a4866d872ee5338b0de/01957724439347680e1b.json +82 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/4ab8140bc7eb4a553d95855c5c2be2cf8c0fbab21b823d76183b6f51e98b6fc5/80038e075c3313d11a1d.json +59 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/4cb7aff9e2a15c151396f2b684013e39d6739f0dec83e5c9dabbfe9d5fcf77b7/e28a26007281cdef7599.json +83 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/6454afdf3e9d66c7226c13a575b718845c25e53b0699600ba2bb4f883e9d841b/b2922eb58f0f3319bfe3.json +63 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/73707b485eab9008c7aba7f5dad0ce2384ac685318d5f888c12fa0d81ed90b19/3aed09186384cc6540ef.json +135 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/7518518c7e077820070186deda960d8cc49db068cdf0ac70664098fa2b6b698c/c7e74a139ce290675aa6.json +65 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/7f05bde17c7b0ffeb657897697f23d182f406b76ced7f1b2cd5741dc93fe2e2e/ca71d7a47e38a69da803.json +126 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/920f44ce6d3e004d1ce547ae06644f7be262180644b04573153aa15d98742edc/d6743412c8cf7830ae4a.json +66 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/7d57cf017fab771c9e95.json +64 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/9fdf2fe50b15951e2b95.json +64 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/c737e0d82a12fca4c663.json +64 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/d139acf64685f15794bb983ff6eb881bdd31304bae88b0ce1ed20a54c21f2265/fcca5755ba87a7768913.json +59 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/gemma3_text/unsloth/gemma-3-270m-it/01957724439347680e1b.json +82 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/granite/hf-internal-testing/tiny-random-GraniteForCausalLM/fcca5755ba87a7768913.json +59 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/idefics3/HuggingFaceTB/SmolVLM-256M-Instruct/2339cde11cbd17accc62.json +189 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/llama/llamafactory/tiny-random-Llama-3/b2922eb58f0f3319bfe3.json +63 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/llama/unsloth/Llama-3.2-1B-Instruct/c737e0d82a12fca4c663.json +64 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/llama4/tiny-random/llama-4/ca71d7a47e38a69da803.json +126 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/mixtral/dacorvo/Mixtral-tiny/80038e075c3313d11a1d.json +59 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/phi3/yujiepan/phi-4-tiny-random/7d7f3f7850d2aa89e3fa.json +60 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/qwen2/Qwen/Qwen2.5-0.5B/e28a26007281cdef7599.json +83 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/qwen2/yujiepan/qwen2.5-128k-tiny-random/c7e74a139ce290675aa6.json +65 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/qwen3_moe/optimum-internal-testing/tiny-random-qwen3_moe/d6743412c8cf7830ae4a.json +66 -0
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/smollm3/HuggingFaceTB/SmolLM3-3B/3aed09186384cc6540ef.json +135 -0
neuronxcc-2.21.33363.0+82129205/MODULE_04c64527506049d877d9+b02446f6/compile_flags.json +1 -0
neuronxcc-2.21.33363.0+82129205/MODULE_04c64527506049d877d9+b02446f6/model.done +0 -0
neuronxcc-2.21.33363.0+82129205/MODULE_04c64527506049d877d9+b02446f6/model.hlo_module.pb +3 -0
neuronxcc-2.21.33363.0+82129205/MODULE_04c64527506049d877d9+b02446f6/model.neff +3 -0
neuronxcc-2.21.33363.0+82129205/MODULE_04c64527506049d877d9+b02446f6/wrapped_neff.hlo +3 -0
neuronxcc-2.21.33363.0+82129205/MODULE_789b4718541dd2474859+a02c3a36/compile_flags.json +1 -0
neuronxcc-2.21.33363.0+82129205/MODULE_789b4718541dd2474859+a02c3a36/model.done +0 -0
neuronxcc-2.21.33363.0+82129205/MODULE_789b4718541dd2474859+a02c3a36/model.hlo_module.pb +3 -0
neuronxcc-2.21.33363.0+82129205/MODULE_789b4718541dd2474859+a02c3a36/model.neff +3 -0
neuronxcc-2.21.33363.0+82129205/MODULE_789b4718541dd2474859+a02c3a36/wrapped_neff.hlo +3 -0
neuronxcc-2.21.33363.0+82129205/MODULE_82382ce0b34b3e272481+6170d8e1/compile_flags.json +1 -0
neuronxcc-2.21.33363.0+82129205/MODULE_82382ce0b34b3e272481+6170d8e1/model.done +0 -0
neuronxcc-2.21.33363.0+82129205/MODULE_82382ce0b34b3e272481+6170d8e1/model.hlo_module.pb +3 -0
neuronxcc-2.21.33363.0+82129205/MODULE_82382ce0b34b3e272481+6170d8e1/model.neff +3 -0

.gitattributes CHANGED Viewed

@@ -17399,3 +17399,8 @@ neuronxcc-2.21.33363.0+82129205/MODULE_880bcc864ca6414a5589+82ae0560/model.neff
 neuronxcc-2.21.33363.0+82129205/MODULE_880bcc864ca6414a5589+82ae0560/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
 neuronxcc-2.21.33363.0+82129205/MODULE_f850a063c4798248da2d+2a75dc25/model.neff filter=lfs diff=lfs merge=lfs -text
 neuronxcc-2.21.33363.0+82129205/MODULE_f850a063c4798248da2d+2a75dc25/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text

 neuronxcc-2.21.33363.0+82129205/MODULE_880bcc864ca6414a5589+82ae0560/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
 neuronxcc-2.21.33363.0+82129205/MODULE_f850a063c4798248da2d+2a75dc25/model.neff filter=lfs diff=lfs merge=lfs -text
 neuronxcc-2.21.33363.0+82129205/MODULE_f850a063c4798248da2d+2a75dc25/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
+neuronxcc-2.21.33363.0+82129205/MODULE_04c64527506049d877d9+b02446f6/model.neff filter=lfs diff=lfs merge=lfs -text
+neuronxcc-2.21.33363.0+82129205/MODULE_04c64527506049d877d9+b02446f6/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
+neuronxcc-2.21.33363.0+82129205/MODULE_789b4718541dd2474859+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
+neuronxcc-2.21.33363.0+82129205/MODULE_789b4718541dd2474859+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
+neuronxcc-2.21.33363.0+82129205/MODULE_82382ce0b34b3e272481+6170d8e1/model.neff filter=lfs diff=lfs merge=lfs -text

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/03b0c107d1cede36875199a5d51decfe04c473de2af9999f8577a028d74d0ab4/2339cde11cbd17accc62.json ADDED Viewed

	@@ -0,0 +1,189 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "HuggingFaceTB/SmolVLM-256M-Instruct",
+  "_task": "image-text-to-text",
+  "architectures": [
+    "Idefics3ForConditionalGeneration"
+  ],
+  "dtype": "bfloat16",
+  "image_token_id": 49190,
+  "model_type": "idefics3",
+  "neuron": {
+    "_serialized_key": "NxDVLMNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "HuggingFaceTB/SmolVLM-256M-Instruct",
+    "checkpoint_revision": "7e3e67edbbed1bf9888184d9df282b700a323964",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": true,
+    "glu_mlp": true,
+    "image_seq_len": 64,
+    "image_size": 512,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 2048,
+    "max_num_images": 17,
+    "max_topk": 256,
+    "n_active_tokens": 2048,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 1024,
+    "sequence_length": 2048,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "bfloat16",
+    "tp_degree": 2
+  },
+  "scale_factor": 4,
+  "text_config": {
+    "_attn_implementation_autoset": false,
+    "_flash_attn_2_enabled": true,
+    "_name_or_path": "None",
+    "architectures": [
+      "VLlama3ForCausalLM"
+    ],
+    "attention_bias": false,
+    "attention_dropout": 0.0,
+    "dtype": "bfloat16",
+    "head_dim": 64,
+    "hidden_act": "silu",
+    "hidden_size": 576,
+    "initializer_range": 0.041666666666666664,
+    "intermediate_size": 1536,
+    "is_llama_config": true,
+    "max_position_embeddings": 8192,
+    "mlp_bias": false,
+    "model_type": "llama",
+    "neftune_noise_alpha": 0.0,
+    "num_attention_heads": 9,
+    "num_hidden_layers": 30,
+    "num_key_value_heads": 3,
+    "pad_token_id": 2,
+    "perceiver_config": {
+      "_attn_implementation_autoset": false,
+      "_name_or_path": "",
+      "add_cross_attention": false,
+      "architectures": null,
+      "attention_dropout": 0.0,
+      "bad_words_ids": null,
+      "begin_suppress_tokens": null,
+      "bos_token_id": null,
+      "chunk_size_feed_forward": 0,
+      "cross_attention_hidden_size": null,
+      "decoder_start_token_id": null,
+      "diversity_penalty": 0.0,
+      "do_sample": false,
+      "early_stopping": false,
+      "encoder_no_repeat_ngram_size": 0,
+      "eos_token_id": null,
+      "exponential_decay_length_penalty": null,
+      "finetuning_task": null,
+      "forced_bos_token_id": null,
+      "forced_eos_token_id": null,
+      "hidden_act": "silu",
+      "id2label": {
+        "0": "LABEL_0",
+        "1": "LABEL_1"
+      },
+      "is_decoder": false,
+      "is_encoder_decoder": false,
+      "label2id": {
+        "LABEL_0": 0,
+        "LABEL_1": 1
+      },
+      "length_penalty": 1.0,
+      "max_length": 20,
+      "min_length": 0,
+      "model_type": "vllama3",
+      "no_repeat_ngram_size": 0,
+      "num_beam_groups": 1,
+      "num_beams": 1,
+      "num_key_value_heads": 1,
+      "num_return_sequences": 1,
+      "output_attentions": false,
+      "output_hidden_states": false,
+      "output_scores": false,
+      "pad_token_id": null,
+      "prefix": null,
+      "problem_type": null,
+      "pruned_heads": {},
+      "qk_layer_norms_perceiver": false,
+      "remove_invalid_values": false,
+      "repetition_penalty": 1.0,
+      "resampler_depth": 6,
+      "resampler_head_dim": 96,
+      "resampler_n_heads": 16,
+      "resampler_n_latents": 64,
+      "return_dict": true,
+      "return_dict_in_generate": false,
+      "sep_token_id": null,
+      "suppress_tokens": null,
+      "task_specific_params": null,
+      "temperature": 1.0,
+      "tf_legacy_loss": false,
+      "tie_encoder_decoder": false,
+      "tie_word_embeddings": true,
+      "tokenizer_class": null,
+      "top_k": 50,
+      "top_p": 1.0,
+      "torch_dtype": null,
+      "torchscript": false,
+      "transformers_version": "4.46.0",
+      "typical_p": 1.0,
+      "use_bfloat16": false
+    },
+    "pixel_shuffle_factor": 4,
+    "pretraining_tp": 1,
+    "qk_layer_norms": false,
+    "rms_norm_eps": 1e-05,
+    "rope_interleaved": false,
+    "rope_scaling": null,
+    "rope_theta": 100000,
+    "transformers.js_config": {
+      "kv_cache_dtype": {
+        "fp16": "float16",
+        "q4f16": "float16"
+      }
+    },
+    "use_cache": true,
+    "use_resampler": false,
+    "vocab_size": 49280
+  },
+  "tie_word_embeddings": false,
+  "transformers.js_config": {
+    "kv_cache_dtype": {
+      "fp16": "float16",
+      "q4f16": "float16"
+    }
+  },
+  "use_cache": true,
+  "vision_config": {
+    "_attn_implementation_autoset": false,
+    "attention_dropout": 0.0,
+    "hidden_act": "gelu_pytorch_tanh",
+    "hidden_size": 768,
+    "image_size": 512,
+    "initializer_range": 0.02,
+    "intermediate_size": 3072,
+    "layer_norm_eps": 1e-06,
+    "max_image_size": {
+      "longest_edge": 512
+    },
+    "model_type": "idefics3_vision",
+    "num_attention_heads": 12,
+    "num_channels": 3,
+    "num_hidden_layers": 12,
+    "patch_size": 16,
+    "size": {
+      "longest_edge": 2048
+    },
+    "tie_word_embeddings": false,
+    "use_base_siglip": true
+  },
+  "vocab_size": 49280
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/2da7a00f0478d50ae1e7f75f085c5b2773b5f355f427c61cf34cb6febd629d96/7d7f3f7850d2aa89e3fa.json ADDED Viewed

	@@ -0,0 +1,60 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "yujiepan/phi-4-tiny-random",
+  "_task": "text-generation",
+  "architectures": [
+    "Phi3ForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "auto_map": {},
+  "dtype": "bfloat16",
+  "embd_pdrop": 0.0,
+  "hidden_act": "silu",
+  "hidden_size": 16,
+  "initializer_range": 0.02,
+  "intermediate_size": 32,
+  "max_position_embeddings": 16384,
+  "model_type": "phi3",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "yujiepan/phi-4-tiny-random",
+    "checkpoint_revision": "18a9a1168dc97ac6d128f811925670c275610f5a",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": true,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "bfloat16",
+    "tp_degree": 2
+  },
+  "num_attention_heads": 2,
+  "num_hidden_layers": 2,
+  "num_key_value_heads": 1,
+  "original_max_position_embeddings": 16384,
+  "partial_rotary_factor": 1.0,
+  "resid_pdrop": 0.0,
+  "rms_norm_eps": 1e-05,
+  "rope_scaling": null,
+  "rope_theta": 250000,
+  "sliding_window": null,
+  "tie_word_embeddings": false,
+  "use_cache": true,
+  "vocab_size": 100352
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/441269935591cad8d370e512c0b93cdd2fce6247c40e5a4866d872ee5338b0de/01957724439347680e1b.json ADDED Viewed

	@@ -0,0 +1,82 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "unsloth/gemma-3-270m-it",
+  "_sliding_window_pattern": 6,
+  "_task": "text-generation",
+  "architectures": [
+    "Gemma3ForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "attn_logit_softcapping": null,
+  "dtype": "bfloat16",
+  "final_logit_softcapping": null,
+  "head_dim": 256,
+  "hidden_activation": "gelu_pytorch_tanh",
+  "hidden_size": 640,
+  "initializer_range": 0.02,
+  "intermediate_size": 2048,
+  "layer_types": [
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention"
+  ],
+  "max_position_embeddings": 32768,
+  "model_type": "gemma3_text",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "unsloth/gemma-3-270m-it",
+    "checkpoint_revision": "23cf460f6bb16954176b3ddcc8d4f250501458a9",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": true,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "bfloat16",
+    "tp_degree": 2
+  },
+  "num_attention_heads": 4,
+  "num_hidden_layers": 18,
+  "num_key_value_heads": 1,
+  "query_pre_attn_scalar": 256,
+  "rms_norm_eps": 1e-06,
+  "rope_local_base_freq": 10000.0,
+  "rope_scaling": null,
+  "rope_theta": 1000000.0,
+  "sliding_window": 512,
+  "unsloth_fixed": true,
+  "use_bidirectional_attention": false,
+  "use_cache": true,
+  "vocab_size": 262144
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/4ab8140bc7eb4a553d95855c5c2be2cf8c0fbab21b823d76183b6f51e98b6fc5/80038e075c3313d11a1d.json ADDED Viewed

	@@ -0,0 +1,59 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "dacorvo/Mixtral-tiny",
+  "_task": "text-generation",
+  "architectures": [
+    "MixtralForCausalLM"
+  ],
+  "attention_dropout": 0.0,
+  "dtype": "float16",
+  "head_dim": 32,
+  "hidden_act": "silu",
+  "hidden_size": 1024,
+  "initializer_range": 0.02,
+  "intermediate_size": 3584,
+  "max_position_embeddings": 1024,
+  "model_type": "mixtral",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "dacorvo/Mixtral-tiny",
+    "checkpoint_revision": "c557ba205ddff6ea911f4719e0d543d6c08356b6",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": false,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": false,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "float16",
+    "tp_degree": 2
+  },
+  "num_attention_heads": 32,
+  "num_experts_per_tok": 2,
+  "num_hidden_layers": 2,
+  "num_key_value_heads": 8,
+  "num_local_experts": 8,
+  "output_router_logits": false,
+  "rms_norm_eps": 1e-05,
+  "rope_theta": 10000.0,
+  "router_aux_loss_coef": 0.001,
+  "router_jitter_noise": 0.0,
+  "sliding_window": 4096,
+  "tie_word_embeddings": false,
+  "use_cache": true,
+  "vocab_size": 32000
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/4cb7aff9e2a15c151396f2b684013e39d6739f0dec83e5c9dabbfe9d5fcf77b7/e28a26007281cdef7599.json ADDED Viewed

	@@ -0,0 +1,83 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "Qwen/Qwen2.5-0.5B",
+  "_task": "text-generation",
+  "architectures": [
+    "Qwen2ForCausalLM"
+  ],
+  "attention_dropout": 0.0,
+  "dtype": "bfloat16",
+  "hidden_act": "silu",
+  "hidden_size": 896,
+  "initializer_range": 0.02,
+  "intermediate_size": 4864,
+  "layer_types": [
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention"
+  ],
+  "max_position_embeddings": 32768,
+  "max_window_layers": 24,
+  "model_type": "qwen2",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 2,
+    "capacity_factor": null,
+    "checkpoint_id": "Qwen/Qwen2.5-0.5B",
+    "checkpoint_revision": "060db6499f32faf8b98477b0a26969ef7d8b9987",
+    "continuous_batching": true,
+    "ep_degree": 1,
+    "fused_qkv": false,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 2,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": false,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "bfloat16",
+    "tp_degree": 2
+  },
+  "num_attention_heads": 14,
+  "num_hidden_layers": 24,
+  "num_key_value_heads": 2,
+  "rms_norm_eps": 1e-06,
+  "rope_scaling": null,
+  "rope_theta": 1000000.0,
+  "sliding_window": null,
+  "tie_word_embeddings": true,
+  "use_cache": true,
+  "use_mrope": false,
+  "use_sliding_window": false,
+  "vocab_size": 151936
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/6454afdf3e9d66c7226c13a575b718845c25e53b0699600ba2bb4f883e9d841b/b2922eb58f0f3319bfe3.json ADDED Viewed

	@@ -0,0 +1,63 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "llamafactory/tiny-random-Llama-3",
+  "_task": "text-generation",
+  "architectures": [
+    "LlamaForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "dtype": "float16",
+  "head_dim": 4,
+  "hidden_act": "silu",
+  "hidden_size": 16,
+  "initializer_range": 0.02,
+  "intermediate_size": 64,
+  "max_position_embeddings": 131072,
+  "mlp_bias": false,
+  "model_type": "llama",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "llamafactory/tiny-random-Llama-3",
+    "checkpoint_revision": "bf2a2e3bf199ad2ee96f02a3c00246c608db22a8",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": true,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "float16",
+    "tp_degree": 2
+  },
+  "num_attention_heads": 4,
+  "num_hidden_layers": 2,
+  "num_key_value_heads": 4,
+  "pretraining_tp": 1,
+  "rms_norm_eps": 1e-05,
+  "rope_scaling": {
+    "factor": 8.0,
+    "high_freq_factor": 4.0,
+    "low_freq_factor": 1.0,
+    "original_max_position_embeddings": 8192,
+    "rope_type": "llama3"
+  },
+  "rope_theta": 500000.0,
+  "tie_word_embeddings": false,
+  "use_cache": true,
+  "vocab_size": 128256
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/73707b485eab9008c7aba7f5dad0ce2384ac685318d5f888c12fa0d81ed90b19/3aed09186384cc6540ef.json ADDED Viewed

	@@ -0,0 +1,135 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "HuggingFaceTB/SmolLM3-3B",
+  "_task": "text-generation",
+  "architectures": [
+    "SmolLM3ForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "dtype": "bfloat16",
+  "hidden_act": "silu",
+  "hidden_size": 2048,
+  "initializer_range": 0.02,
+  "intermediate_size": 11008,
+  "layer_types": [
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention"
+  ],
+  "max_position_embeddings": 65536,
+  "max_window_layers": 28,
+  "mlp_bias": false,
+  "model_type": "smollm3",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "HuggingFaceTB/SmolLM3-3B",
+    "checkpoint_revision": "a07cc9a04f16550a088caea529712d1d335b0ac1",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": true,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "bfloat16",
+    "tp_degree": 2
+  },
+  "no_rope_layer_interval": 4,
+  "no_rope_layers": [
+    1,
+    1,
+    1,
+    0,
+    1,
+    1,
+    1,
+    0,
+    1,
+    1,
+    1,
+    0,
+    1,
+    1,
+    1,
+    0,
+    1,
+    1,
+    1,
+    0,
+    1,
+    1,
+    1,
+    0,
+    1,
+    1,
+    1,
+    0,
+    1,
+    1,
+    1,
+    0,
+    1,
+    1,
+    1,
+    0
+  ],
+  "num_attention_heads": 16,
+  "num_hidden_layers": 36,
+  "num_key_value_heads": 4,
+  "pretraining_tp": 2,
+  "rms_norm_eps": 1e-06,
+  "rope_scaling": null,
+  "rope_theta": 5000000.0,
+  "sliding_window": null,
+  "use_cache": false,
+  "use_sliding_window": false,
+  "vocab_size": 128256
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/7518518c7e077820070186deda960d8cc49db068cdf0ac70664098fa2b6b698c/c7e74a139ce290675aa6.json ADDED Viewed

	@@ -0,0 +1,65 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "yujiepan/qwen2.5-128k-tiny-random",
+  "_task": "text-generation",
+  "architectures": [
+    "Qwen2ForCausalLM"
+  ],
+  "attention_dropout": 0.0,
+  "dtype": "bfloat16",
+  "hidden_act": "silu",
+  "hidden_size": 8,
+  "initializer_range": 0.02,
+  "intermediate_size": 16,
+  "layer_types": [
+    "full_attention",
+    "full_attention"
+  ],
+  "max_position_embeddings": 32768,
+  "max_window_layers": 1,
+  "model_type": "qwen2",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "yujiepan/qwen2.5-128k-tiny-random",
+    "checkpoint_revision": "c8296d4ca3f87782876d2382fbb6481d1beb8ef0",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": false,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "bfloat16",
+    "tp_degree": 2
+  },
+  "num_attention_heads": 4,
+  "num_hidden_layers": 2,
+  "num_key_value_heads": 2,
+  "rms_norm_eps": 1e-06,
+  "rope_scaling": {
+    "factor": 4.0,
+    "original_max_position_embeddings": 32768,
+    "rope_type": "yarn",
+    "type": "yarn"
+  },
+  "rope_theta": 1000000.0,
+  "sliding_window": null,
+  "tie_word_embeddings": false,
+  "use_cache": true,
+  "use_sliding_window": false,
+  "vocab_size": 152064
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/7f05bde17c7b0ffeb657897697f23d182f406b76ced7f1b2cd5741dc93fe2e2e/ca71d7a47e38a69da803.json ADDED Viewed

	@@ -0,0 +1,126 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "tiny-random/llama-4",
+  "_task": "text-generation",
+  "architectures": [
+    "Llama4ForConditionalGeneration"
+  ],
+  "boi_token_index": 200080,
+  "dtype": "bfloat16",
+  "eoi_token_index": 200081,
+  "image_token_index": 200092,
+  "model_type": "llama4",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "tiny-random/llama-4",
+    "checkpoint_revision": "9e716f5d4d1ffe0a44a15f46f4a12b840439aba4",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": false,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "bfloat16",
+    "tp_degree": 2
+  },
+  "text_config": {
+    "_attn_implementation_autoset": true,
+    "attention_bias": false,
+    "attention_chunk_size": 128,
+    "attention_dropout": 0.0,
+    "attn_scale": 0.1,
+    "attn_temperature_tuning": 4,
+    "bos_token_id": 200000,
+    "cache_implementation": "hybrid",
+    "dtype": "bfloat16",
+    "eos_token_id": [
+      200001,
+      200007,
+      200008
+    ],
+    "floor_scale": 8192,
+    "for_llm_compressor": false,
+    "head_dim": 32,
+    "hidden_act": "silu",
+    "hidden_size": 32,
+    "initializer_range": 0.02,
+    "interleave_moe_layer_step": 2,
+    "intermediate_size": 64,
+    "intermediate_size_mlp": 128,
+    "layer_types": [
+      "chunked_attention",
+      "chunked_attention",
+      "chunked_attention",
+      "full_attention"
+    ],
+    "max_position_embeddings": 1048576,
+    "model_type": "llama4_text",
+    "moe_layers": [
+      1,
+      3
+    ],
+    "no_rope_layers": [
+      1,
+      1,
+      1,
+      0
+    ],
+    "num_attention_heads": 1,
+    "num_experts_per_tok": 1,
+    "num_hidden_layers": 4,
+    "num_key_value_heads": 1,
+    "num_local_experts": 8,
+    "output_router_logits": false,
+    "pad_token_id": 200018,
+    "rms_norm_eps": 1e-05,
+    "rope_scaling": null,
+    "rope_theta": 500000.0,
+    "router_aux_loss_coef": 0.001,
+    "router_jitter_noise": 0.0,
+    "tie_word_embeddings": true,
+    "use_cache": true,
+    "use_qk_norm": true,
+    "vocab_size": 202048
+  },
+  "tie_word_embeddings": false,
+  "vision_config": {
+    "_attn_implementation_autoset": true,
+    "_vision_feature_layer": -1,
+    "attention_dropout": 0.0,
+    "hidden_act": "gelu",
+    "hidden_size": 32,
+    "image_size": 336,
+    "initializer_range": 0.02,
+    "intermediate_size": 128,
+    "model_type": "llama4_vision_model",
+    "multi_modal_projector_bias": false,
+    "norm_eps": 1e-05,
+    "num_attention_heads": 1,
+    "num_channels": 3,
+    "num_hidden_layers": 2,
+    "patch_size": 14,
+    "pixel_shuffle_ratio": 0.5,
+    "projector_dropout": 0.0,
+    "projector_input_dim": 32,
+    "projector_output_dim": 32,
+    "rope_theta": 10000,
+    "vision_feature_layer": -1,
+    "vision_feature_select_strategy": "default",
+    "vision_output_dim": 32
+  }
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/920f44ce6d3e004d1ce547ae06644f7be262180644b04573153aa15d98742edc/d6743412c8cf7830ae4a.json ADDED Viewed

	@@ -0,0 +1,66 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "optimum-internal-testing/tiny-random-qwen3_moe",
+  "_task": "text-generation",
+  "architectures": [
+    "Qwen3MoeForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "decoder_sparse_step": 2,
+  "dtype": "float32",
+  "head_dim": 32,
+  "hidden_act": "silu",
+  "hidden_size": 64,
+  "initializer_range": 0.02,
+  "intermediate_size": 128,
+  "max_position_embeddings": 40960,
+  "max_window_layers": 1,
+  "mlp_only_layers": [],
+  "model_type": "qwen3_moe",
+  "moe_intermediate_size": 128,
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "optimum-internal-testing/tiny-random-qwen3_moe",
+    "checkpoint_revision": "e0230be2839556b44b7400a233c73c74b4abb7af",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": false,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "float32",
+    "tp_degree": 2
+  },
+  "norm_topk_prob": true,
+  "num_attention_heads": 2,
+  "num_experts": 8,
+  "num_experts_per_tok": 2,
+  "num_hidden_layers": 2,
+  "num_key_value_heads": 1,
+  "output_router_logits": false,
+  "rms_norm_eps": 1e-06,
+  "rope_scaling": null,
+  "rope_theta": 1000000.0,
+  "router_aux_loss_coef": 0.001,
+  "sliding_window": null,
+  "tie_word_embeddings": true,
+  "use_cache": true,
+  "use_sliding_window": false,
+  "vocab_size": 151936
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/7d57cf017fab771c9e95.json ADDED Viewed

	@@ -0,0 +1,64 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "unsloth/Llama-3.2-1B-Instruct",
+  "_task": "text-generation",
+  "architectures": [
+    "LlamaForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "dtype": "bfloat16",
+  "head_dim": 64,
+  "hidden_act": "silu",
+  "hidden_size": 2048,
+  "initializer_range": 0.02,
+  "intermediate_size": 8192,
+  "max_position_embeddings": 131072,
+  "mlp_bias": false,
+  "model_type": "llama",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "unsloth/Llama-3.2-1B-Instruct",
+    "checkpoint_revision": "5a8abab4a5d6f164389b1079fb721cfab8d7126c",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": true,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 4096,
+    "max_topk": 256,
+    "n_active_tokens": 4096,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 1024,
+    "sequence_length": 4096,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "bfloat16",
+    "tp_degree": 2
+  },
+  "num_attention_heads": 32,
+  "num_hidden_layers": 16,
+  "num_key_value_heads": 8,
+  "pretraining_tp": 1,
+  "rms_norm_eps": 1e-05,
+  "rope_scaling": {
+    "factor": 32.0,
+    "high_freq_factor": 4.0,
+    "low_freq_factor": 1.0,
+    "original_max_position_embeddings": 8192,
+    "rope_type": "llama3"
+  },
+  "rope_theta": 500000.0,
+  "tie_word_embeddings": true,
+  "unsloth_fixed": true,
+  "use_cache": true,
+  "vocab_size": 128256
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/9fdf2fe50b15951e2b95.json ADDED Viewed

	@@ -0,0 +1,64 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "unsloth/Llama-3.2-1B-Instruct",
+  "_task": "text-generation",
+  "architectures": [
+    "LlamaForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "dtype": "bfloat16",
+  "head_dim": 64,
+  "hidden_act": "silu",
+  "hidden_size": 2048,
+  "initializer_range": 0.02,
+  "intermediate_size": 8192,
+  "max_position_embeddings": 131072,
+  "mlp_bias": false,
+  "model_type": "llama",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "unsloth/Llama-3.2-1B-Instruct",
+    "checkpoint_revision": null,
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": false,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 4096,
+    "max_topk": 256,
+    "n_active_tokens": 4096,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": false,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 4096,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "bfloat16",
+    "tp_degree": 2
+  },
+  "num_attention_heads": 32,
+  "num_hidden_layers": 16,
+  "num_key_value_heads": 8,
+  "pretraining_tp": 1,
+  "rms_norm_eps": 1e-05,
+  "rope_scaling": {
+    "factor": 32.0,
+    "high_freq_factor": 4.0,
+    "low_freq_factor": 1.0,
+    "original_max_position_embeddings": 8192,
+    "rope_type": "llama3"
+  },
+  "rope_theta": 500000.0,
+  "tie_word_embeddings": true,
+  "unsloth_fixed": true,
+  "use_cache": true,
+  "vocab_size": 128256
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/c737e0d82a12fca4c663.json ADDED Viewed

	@@ -0,0 +1,64 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "unsloth/Llama-3.2-1B-Instruct",
+  "_task": "text-generation",
+  "architectures": [
+    "LlamaForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "dtype": "bfloat16",
+  "head_dim": 64,
+  "hidden_act": "silu",
+  "hidden_size": 2048,
+  "initializer_range": 0.02,
+  "intermediate_size": 8192,
+  "max_position_embeddings": 131072,
+  "mlp_bias": false,
+  "model_type": "llama",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "unsloth/Llama-3.2-1B-Instruct",
+    "checkpoint_revision": null,
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": false,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 4096,
+    "max_topk": 256,
+    "n_active_tokens": 4096,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": false,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 4096,
+    "speculation_length": 5,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "bfloat16",
+    "tp_degree": 2
+  },
+  "num_attention_heads": 32,
+  "num_hidden_layers": 16,
+  "num_key_value_heads": 8,
+  "pretraining_tp": 1,
+  "rms_norm_eps": 1e-05,
+  "rope_scaling": {
+    "factor": 32.0,
+    "high_freq_factor": 4.0,
+    "low_freq_factor": 1.0,
+    "original_max_position_embeddings": 8192,
+    "rope_type": "llama3"
+  },
+  "rope_theta": 500000.0,
+  "tie_word_embeddings": true,
+  "unsloth_fixed": true,
+  "use_cache": true,
+  "vocab_size": 128256
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/d139acf64685f15794bb983ff6eb881bdd31304bae88b0ce1ed20a54c21f2265/fcca5755ba87a7768913.json ADDED Viewed

	@@ -0,0 +1,59 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "hf-internal-testing/tiny-random-GraniteForCausalLM",
+  "_task": "text-generation",
+  "architectures": [
+    "GraniteForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "attention_multiplier": 1.0,
+  "dtype": "float32",
+  "embedding_multiplier": 1.0,
+  "hidden_act": "silu",
+  "hidden_size": 32,
+  "initializer_range": 0.02,
+  "intermediate_size": 64,
+  "logits_scaling": 1.0,
+  "max_position_embeddings": 2048,
+  "mlp_bias": false,
+  "model_type": "granite",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "hf-internal-testing/tiny-random-GraniteForCausalLM",
+    "checkpoint_revision": "c3074ebc0ac2fe545305f5e5f6cce2cc9b2aa0c5",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": true,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "float32",
+    "tp_degree": 2
+  },
+  "num_attention_heads": 4,
+  "num_hidden_layers": 2,
+  "num_key_value_heads": 4,
+  "residual_multiplier": 1.0,
+  "rms_norm_eps": 1e-06,
+  "rope_scaling": null,
+  "rope_theta": 10000.0,
+  "tie_word_embeddings": false,
+  "use_cache": true,
+  "vocab_size": 49152
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/gemma3_text/unsloth/gemma-3-270m-it/01957724439347680e1b.json ADDED Viewed

	@@ -0,0 +1,82 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "unsloth/gemma-3-270m-it",
+  "_sliding_window_pattern": 6,
+  "_task": "text-generation",
+  "architectures": [
+    "Gemma3ForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "attn_logit_softcapping": null,
+  "dtype": "bfloat16",
+  "final_logit_softcapping": null,
+  "head_dim": 256,
+  "hidden_activation": "gelu_pytorch_tanh",
+  "hidden_size": 640,
+  "initializer_range": 0.02,
+  "intermediate_size": 2048,
+  "layer_types": [
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "sliding_attention",
+    "full_attention"
+  ],
+  "max_position_embeddings": 32768,
+  "model_type": "gemma3_text",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "unsloth/gemma-3-270m-it",
+    "checkpoint_revision": "23cf460f6bb16954176b3ddcc8d4f250501458a9",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": true,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "bfloat16",
+    "tp_degree": 2
+  },
+  "num_attention_heads": 4,
+  "num_hidden_layers": 18,
+  "num_key_value_heads": 1,
+  "query_pre_attn_scalar": 256,
+  "rms_norm_eps": 1e-06,
+  "rope_local_base_freq": 10000.0,
+  "rope_scaling": null,
+  "rope_theta": 1000000.0,
+  "sliding_window": 512,
+  "unsloth_fixed": true,
+  "use_bidirectional_attention": false,
+  "use_cache": true,
+  "vocab_size": 262144
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/granite/hf-internal-testing/tiny-random-GraniteForCausalLM/fcca5755ba87a7768913.json ADDED Viewed

	@@ -0,0 +1,59 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "hf-internal-testing/tiny-random-GraniteForCausalLM",
+  "_task": "text-generation",
+  "architectures": [
+    "GraniteForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "attention_multiplier": 1.0,
+  "dtype": "float32",
+  "embedding_multiplier": 1.0,
+  "hidden_act": "silu",
+  "hidden_size": 32,
+  "initializer_range": 0.02,
+  "intermediate_size": 64,
+  "logits_scaling": 1.0,
+  "max_position_embeddings": 2048,
+  "mlp_bias": false,
+  "model_type": "granite",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "hf-internal-testing/tiny-random-GraniteForCausalLM",
+    "checkpoint_revision": "c3074ebc0ac2fe545305f5e5f6cce2cc9b2aa0c5",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": true,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "float32",
+    "tp_degree": 2
+  },
+  "num_attention_heads": 4,
+  "num_hidden_layers": 2,
+  "num_key_value_heads": 4,
+  "residual_multiplier": 1.0,
+  "rms_norm_eps": 1e-06,
+  "rope_scaling": null,
+  "rope_theta": 10000.0,
+  "tie_word_embeddings": false,
+  "use_cache": true,
+  "vocab_size": 49152
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/idefics3/HuggingFaceTB/SmolVLM-256M-Instruct/2339cde11cbd17accc62.json ADDED Viewed

	@@ -0,0 +1,189 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "HuggingFaceTB/SmolVLM-256M-Instruct",
+  "_task": "image-text-to-text",
+  "architectures": [
+    "Idefics3ForConditionalGeneration"
+  ],
+  "dtype": "bfloat16",
+  "image_token_id": 49190,
+  "model_type": "idefics3",
+  "neuron": {
+    "_serialized_key": "NxDVLMNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "HuggingFaceTB/SmolVLM-256M-Instruct",
+    "checkpoint_revision": "7e3e67edbbed1bf9888184d9df282b700a323964",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": true,
+    "glu_mlp": true,
+    "image_seq_len": 64,
+    "image_size": 512,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 2048,
+    "max_num_images": 17,
+    "max_topk": 256,
+    "n_active_tokens": 2048,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 1024,
+    "sequence_length": 2048,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "bfloat16",
+    "tp_degree": 2
+  },
+  "scale_factor": 4,
+  "text_config": {
+    "_attn_implementation_autoset": false,
+    "_flash_attn_2_enabled": true,
+    "_name_or_path": "None",
+    "architectures": [
+      "VLlama3ForCausalLM"
+    ],
+    "attention_bias": false,
+    "attention_dropout": 0.0,
+    "dtype": "bfloat16",
+    "head_dim": 64,
+    "hidden_act": "silu",
+    "hidden_size": 576,
+    "initializer_range": 0.041666666666666664,
+    "intermediate_size": 1536,
+    "is_llama_config": true,
+    "max_position_embeddings": 8192,
+    "mlp_bias": false,
+    "model_type": "llama",
+    "neftune_noise_alpha": 0.0,
+    "num_attention_heads": 9,
+    "num_hidden_layers": 30,
+    "num_key_value_heads": 3,
+    "pad_token_id": 2,
+    "perceiver_config": {
+      "_attn_implementation_autoset": false,
+      "_name_or_path": "",
+      "add_cross_attention": false,
+      "architectures": null,
+      "attention_dropout": 0.0,
+      "bad_words_ids": null,
+      "begin_suppress_tokens": null,
+      "bos_token_id": null,
+      "chunk_size_feed_forward": 0,
+      "cross_attention_hidden_size": null,
+      "decoder_start_token_id": null,
+      "diversity_penalty": 0.0,
+      "do_sample": false,
+      "early_stopping": false,
+      "encoder_no_repeat_ngram_size": 0,
+      "eos_token_id": null,
+      "exponential_decay_length_penalty": null,
+      "finetuning_task": null,
+      "forced_bos_token_id": null,
+      "forced_eos_token_id": null,
+      "hidden_act": "silu",
+      "id2label": {
+        "0": "LABEL_0",
+        "1": "LABEL_1"
+      },
+      "is_decoder": false,
+      "is_encoder_decoder": false,
+      "label2id": {
+        "LABEL_0": 0,
+        "LABEL_1": 1
+      },
+      "length_penalty": 1.0,
+      "max_length": 20,
+      "min_length": 0,
+      "model_type": "vllama3",
+      "no_repeat_ngram_size": 0,
+      "num_beam_groups": 1,
+      "num_beams": 1,
+      "num_key_value_heads": 1,
+      "num_return_sequences": 1,
+      "output_attentions": false,
+      "output_hidden_states": false,
+      "output_scores": false,
+      "pad_token_id": null,
+      "prefix": null,
+      "problem_type": null,
+      "pruned_heads": {},
+      "qk_layer_norms_perceiver": false,
+      "remove_invalid_values": false,
+      "repetition_penalty": 1.0,
+      "resampler_depth": 6,
+      "resampler_head_dim": 96,
+      "resampler_n_heads": 16,
+      "resampler_n_latents": 64,
+      "return_dict": true,
+      "return_dict_in_generate": false,
+      "sep_token_id": null,
+      "suppress_tokens": null,
+      "task_specific_params": null,
+      "temperature": 1.0,
+      "tf_legacy_loss": false,
+      "tie_encoder_decoder": false,
+      "tie_word_embeddings": true,
+      "tokenizer_class": null,
+      "top_k": 50,
+      "top_p": 1.0,
+      "torch_dtype": null,
+      "torchscript": false,
+      "transformers_version": "4.46.0",
+      "typical_p": 1.0,
+      "use_bfloat16": false
+    },
+    "pixel_shuffle_factor": 4,
+    "pretraining_tp": 1,
+    "qk_layer_norms": false,
+    "rms_norm_eps": 1e-05,
+    "rope_interleaved": false,
+    "rope_scaling": null,
+    "rope_theta": 100000,
+    "transformers.js_config": {
+      "kv_cache_dtype": {
+        "fp16": "float16",
+        "q4f16": "float16"
+      }
+    },
+    "use_cache": true,
+    "use_resampler": false,
+    "vocab_size": 49280
+  },
+  "tie_word_embeddings": false,
+  "transformers.js_config": {
+    "kv_cache_dtype": {
+      "fp16": "float16",
+      "q4f16": "float16"
+    }
+  },
+  "use_cache": true,
+  "vision_config": {
+    "_attn_implementation_autoset": false,
+    "attention_dropout": 0.0,
+    "hidden_act": "gelu_pytorch_tanh",
+    "hidden_size": 768,
+    "image_size": 512,
+    "initializer_range": 0.02,
+    "intermediate_size": 3072,
+    "layer_norm_eps": 1e-06,
+    "max_image_size": {
+      "longest_edge": 512
+    },
+    "model_type": "idefics3_vision",
+    "num_attention_heads": 12,
+    "num_channels": 3,
+    "num_hidden_layers": 12,
+    "patch_size": 16,
+    "size": {
+      "longest_edge": 2048
+    },
+    "tie_word_embeddings": false,
+    "use_base_siglip": true
+  },
+  "vocab_size": 49280
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/llama/llamafactory/tiny-random-Llama-3/b2922eb58f0f3319bfe3.json ADDED Viewed

	@@ -0,0 +1,63 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "llamafactory/tiny-random-Llama-3",
+  "_task": "text-generation",
+  "architectures": [
+    "LlamaForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "dtype": "float16",
+  "head_dim": 4,
+  "hidden_act": "silu",
+  "hidden_size": 16,
+  "initializer_range": 0.02,
+  "intermediate_size": 64,
+  "max_position_embeddings": 131072,
+  "mlp_bias": false,
+  "model_type": "llama",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "llamafactory/tiny-random-Llama-3",
+    "checkpoint_revision": "bf2a2e3bf199ad2ee96f02a3c00246c608db22a8",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": true,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "float16",
+    "tp_degree": 2
+  },
+  "num_attention_heads": 4,
+  "num_hidden_layers": 2,
+  "num_key_value_heads": 4,
+  "pretraining_tp": 1,
+  "rms_norm_eps": 1e-05,
+  "rope_scaling": {
+    "factor": 8.0,
+    "high_freq_factor": 4.0,
+    "low_freq_factor": 1.0,
+    "original_max_position_embeddings": 8192,
+    "rope_type": "llama3"
+  },
+  "rope_theta": 500000.0,
+  "tie_word_embeddings": false,
+  "use_cache": true,
+  "vocab_size": 128256
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/llama/unsloth/Llama-3.2-1B-Instruct/c737e0d82a12fca4c663.json ADDED Viewed

	@@ -0,0 +1,64 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "unsloth/Llama-3.2-1B-Instruct",
+  "_task": "text-generation",
+  "architectures": [
+    "LlamaForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "dtype": "bfloat16",
+  "head_dim": 64,
+  "hidden_act": "silu",
+  "hidden_size": 2048,
+  "initializer_range": 0.02,
+  "intermediate_size": 8192,
+  "max_position_embeddings": 131072,
+  "mlp_bias": false,
+  "model_type": "llama",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "unsloth/Llama-3.2-1B-Instruct",
+    "checkpoint_revision": null,
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": false,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 4096,
+    "max_topk": 256,
+    "n_active_tokens": 4096,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": false,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 4096,
+    "speculation_length": 5,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "bfloat16",
+    "tp_degree": 2
+  },
+  "num_attention_heads": 32,
+  "num_hidden_layers": 16,
+  "num_key_value_heads": 8,
+  "pretraining_tp": 1,
+  "rms_norm_eps": 1e-05,
+  "rope_scaling": {
+    "factor": 32.0,
+    "high_freq_factor": 4.0,
+    "low_freq_factor": 1.0,
+    "original_max_position_embeddings": 8192,
+    "rope_type": "llama3"
+  },
+  "rope_theta": 500000.0,
+  "tie_word_embeddings": true,
+  "unsloth_fixed": true,
+  "use_cache": true,
+  "vocab_size": 128256
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/llama4/tiny-random/llama-4/ca71d7a47e38a69da803.json ADDED Viewed

	@@ -0,0 +1,126 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "tiny-random/llama-4",
+  "_task": "text-generation",
+  "architectures": [
+    "Llama4ForConditionalGeneration"
+  ],
+  "boi_token_index": 200080,
+  "dtype": "bfloat16",
+  "eoi_token_index": 200081,
+  "image_token_index": 200092,
+  "model_type": "llama4",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "tiny-random/llama-4",
+    "checkpoint_revision": "9e716f5d4d1ffe0a44a15f46f4a12b840439aba4",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": false,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "bfloat16",
+    "tp_degree": 2
+  },
+  "text_config": {
+    "_attn_implementation_autoset": true,
+    "attention_bias": false,
+    "attention_chunk_size": 128,
+    "attention_dropout": 0.0,
+    "attn_scale": 0.1,
+    "attn_temperature_tuning": 4,
+    "bos_token_id": 200000,
+    "cache_implementation": "hybrid",
+    "dtype": "bfloat16",
+    "eos_token_id": [
+      200001,
+      200007,
+      200008
+    ],
+    "floor_scale": 8192,
+    "for_llm_compressor": false,
+    "head_dim": 32,
+    "hidden_act": "silu",
+    "hidden_size": 32,
+    "initializer_range": 0.02,
+    "interleave_moe_layer_step": 2,
+    "intermediate_size": 64,
+    "intermediate_size_mlp": 128,
+    "layer_types": [
+      "chunked_attention",
+      "chunked_attention",
+      "chunked_attention",
+      "full_attention"
+    ],
+    "max_position_embeddings": 1048576,
+    "model_type": "llama4_text",
+    "moe_layers": [
+      1,
+      3
+    ],
+    "no_rope_layers": [
+      1,
+      1,
+      1,
+      0
+    ],
+    "num_attention_heads": 1,
+    "num_experts_per_tok": 1,
+    "num_hidden_layers": 4,
+    "num_key_value_heads": 1,
+    "num_local_experts": 8,
+    "output_router_logits": false,
+    "pad_token_id": 200018,
+    "rms_norm_eps": 1e-05,
+    "rope_scaling": null,
+    "rope_theta": 500000.0,
+    "router_aux_loss_coef": 0.001,
+    "router_jitter_noise": 0.0,
+    "tie_word_embeddings": true,
+    "use_cache": true,
+    "use_qk_norm": true,
+    "vocab_size": 202048
+  },
+  "tie_word_embeddings": false,
+  "vision_config": {
+    "_attn_implementation_autoset": true,
+    "_vision_feature_layer": -1,
+    "attention_dropout": 0.0,
+    "hidden_act": "gelu",
+    "hidden_size": 32,
+    "image_size": 336,
+    "initializer_range": 0.02,
+    "intermediate_size": 128,
+    "model_type": "llama4_vision_model",
+    "multi_modal_projector_bias": false,
+    "norm_eps": 1e-05,
+    "num_attention_heads": 1,
+    "num_channels": 3,
+    "num_hidden_layers": 2,
+    "patch_size": 14,
+    "pixel_shuffle_ratio": 0.5,
+    "projector_dropout": 0.0,
+    "projector_input_dim": 32,
+    "projector_output_dim": 32,
+    "rope_theta": 10000,
+    "vision_feature_layer": -1,
+    "vision_feature_select_strategy": "default",
+    "vision_output_dim": 32
+  }
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/mixtral/dacorvo/Mixtral-tiny/80038e075c3313d11a1d.json ADDED Viewed

	@@ -0,0 +1,59 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "dacorvo/Mixtral-tiny",
+  "_task": "text-generation",
+  "architectures": [
+    "MixtralForCausalLM"
+  ],
+  "attention_dropout": 0.0,
+  "dtype": "float16",
+  "head_dim": 32,
+  "hidden_act": "silu",
+  "hidden_size": 1024,
+  "initializer_range": 0.02,
+  "intermediate_size": 3584,
+  "max_position_embeddings": 1024,
+  "model_type": "mixtral",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "dacorvo/Mixtral-tiny",
+    "checkpoint_revision": "c557ba205ddff6ea911f4719e0d543d6c08356b6",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": false,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": false,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "float16",
+    "tp_degree": 2
+  },
+  "num_attention_heads": 32,
+  "num_experts_per_tok": 2,
+  "num_hidden_layers": 2,
+  "num_key_value_heads": 8,
+  "num_local_experts": 8,
+  "output_router_logits": false,
+  "rms_norm_eps": 1e-05,
+  "rope_theta": 10000.0,
+  "router_aux_loss_coef": 0.001,
+  "router_jitter_noise": 0.0,
+  "sliding_window": 4096,
+  "tie_word_embeddings": false,
+  "use_cache": true,
+  "vocab_size": 32000
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/phi3/yujiepan/phi-4-tiny-random/7d7f3f7850d2aa89e3fa.json ADDED Viewed

	@@ -0,0 +1,60 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "yujiepan/phi-4-tiny-random",
+  "_task": "text-generation",
+  "architectures": [
+    "Phi3ForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "auto_map": {},
+  "dtype": "bfloat16",
+  "embd_pdrop": 0.0,
+  "hidden_act": "silu",
+  "hidden_size": 16,
+  "initializer_range": 0.02,
+  "intermediate_size": 32,
+  "max_position_embeddings": 16384,
+  "model_type": "phi3",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "yujiepan/phi-4-tiny-random",
+    "checkpoint_revision": "18a9a1168dc97ac6d128f811925670c275610f5a",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": true,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "bfloat16",
+    "tp_degree": 2
+  },
+  "num_attention_heads": 2,
+  "num_hidden_layers": 2,
+  "num_key_value_heads": 1,
+  "original_max_position_embeddings": 16384,
+  "partial_rotary_factor": 1.0,
+  "resid_pdrop": 0.0,
+  "rms_norm_eps": 1e-05,
+  "rope_scaling": null,
+  "rope_theta": 250000,
+  "sliding_window": null,
+  "tie_word_embeddings": false,
+  "use_cache": true,
+  "vocab_size": 100352
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/qwen2/Qwen/Qwen2.5-0.5B/e28a26007281cdef7599.json ADDED Viewed

	@@ -0,0 +1,83 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "Qwen/Qwen2.5-0.5B",
+  "_task": "text-generation",
+  "architectures": [
+    "Qwen2ForCausalLM"
+  ],
+  "attention_dropout": 0.0,
+  "dtype": "bfloat16",
+  "hidden_act": "silu",
+  "hidden_size": 896,
+  "initializer_range": 0.02,
+  "intermediate_size": 4864,
+  "layer_types": [
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention"
+  ],
+  "max_position_embeddings": 32768,
+  "max_window_layers": 24,
+  "model_type": "qwen2",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 2,
+    "capacity_factor": null,
+    "checkpoint_id": "Qwen/Qwen2.5-0.5B",
+    "checkpoint_revision": "060db6499f32faf8b98477b0a26969ef7d8b9987",
+    "continuous_batching": true,
+    "ep_degree": 1,
+    "fused_qkv": false,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 2,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": false,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "bfloat16",
+    "tp_degree": 2
+  },
+  "num_attention_heads": 14,
+  "num_hidden_layers": 24,
+  "num_key_value_heads": 2,
+  "rms_norm_eps": 1e-06,
+  "rope_scaling": null,
+  "rope_theta": 1000000.0,
+  "sliding_window": null,
+  "tie_word_embeddings": true,
+  "use_cache": true,
+  "use_mrope": false,
+  "use_sliding_window": false,
+  "vocab_size": 151936
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/qwen2/yujiepan/qwen2.5-128k-tiny-random/c7e74a139ce290675aa6.json ADDED Viewed

	@@ -0,0 +1,65 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "yujiepan/qwen2.5-128k-tiny-random",
+  "_task": "text-generation",
+  "architectures": [
+    "Qwen2ForCausalLM"
+  ],
+  "attention_dropout": 0.0,
+  "dtype": "bfloat16",
+  "hidden_act": "silu",
+  "hidden_size": 8,
+  "initializer_range": 0.02,
+  "intermediate_size": 16,
+  "layer_types": [
+    "full_attention",
+    "full_attention"
+  ],
+  "max_position_embeddings": 32768,
+  "max_window_layers": 1,
+  "model_type": "qwen2",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "yujiepan/qwen2.5-128k-tiny-random",
+    "checkpoint_revision": "c8296d4ca3f87782876d2382fbb6481d1beb8ef0",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": false,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "bfloat16",
+    "tp_degree": 2
+  },
+  "num_attention_heads": 4,
+  "num_hidden_layers": 2,
+  "num_key_value_heads": 2,
+  "rms_norm_eps": 1e-06,
+  "rope_scaling": {
+    "factor": 4.0,
+    "original_max_position_embeddings": 32768,
+    "rope_type": "yarn",
+    "type": "yarn"
+  },
+  "rope_theta": 1000000.0,
+  "sliding_window": null,
+  "tie_word_embeddings": false,
+  "use_cache": true,
+  "use_sliding_window": false,
+  "vocab_size": 152064
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/qwen3_moe/optimum-internal-testing/tiny-random-qwen3_moe/d6743412c8cf7830ae4a.json ADDED Viewed

	@@ -0,0 +1,66 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "optimum-internal-testing/tiny-random-qwen3_moe",
+  "_task": "text-generation",
+  "architectures": [
+    "Qwen3MoeForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "decoder_sparse_step": 2,
+  "dtype": "float32",
+  "head_dim": 32,
+  "hidden_act": "silu",
+  "hidden_size": 64,
+  "initializer_range": 0.02,
+  "intermediate_size": 128,
+  "max_position_embeddings": 40960,
+  "max_window_layers": 1,
+  "mlp_only_layers": [],
+  "model_type": "qwen3_moe",
+  "moe_intermediate_size": 128,
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "optimum-internal-testing/tiny-random-qwen3_moe",
+    "checkpoint_revision": "e0230be2839556b44b7400a233c73c74b4abb7af",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": false,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "float32",
+    "tp_degree": 2
+  },
+  "norm_topk_prob": true,
+  "num_attention_heads": 2,
+  "num_experts": 8,
+  "num_experts_per_tok": 2,
+  "num_hidden_layers": 2,
+  "num_key_value_heads": 1,
+  "output_router_logits": false,
+  "rms_norm_eps": 1e-06,
+  "rope_scaling": null,
+  "rope_theta": 1000000.0,
+  "router_aux_loss_coef": 0.001,
+  "sliding_window": null,
+  "tie_word_embeddings": true,
+  "use_cache": true,
+  "use_sliding_window": false,
+  "vocab_size": 151936
+}

neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev4/smollm3/HuggingFaceTB/SmolLM3-3B/3aed09186384cc6540ef.json ADDED Viewed

	@@ -0,0 +1,135 @@

+{
+  "_entry_class": "SingleModelCacheEntry",
+  "_model_id": "HuggingFaceTB/SmolLM3-3B",
+  "_task": "text-generation",
+  "architectures": [
+    "SmolLM3ForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "dtype": "bfloat16",
+  "hidden_act": "silu",
+  "hidden_size": 2048,
+  "initializer_range": 0.02,
+  "intermediate_size": 11008,
+  "layer_types": [
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention",
+    "full_attention"
+  ],
+  "max_position_embeddings": 65536,
+  "max_window_layers": 28,
+  "mlp_bias": false,
+  "model_type": "smollm3",
+  "neuron": {
+    "_serialized_key": "NxDNeuronConfig",
+    "batch_size": 1,
+    "capacity_factor": null,
+    "checkpoint_id": "HuggingFaceTB/SmolLM3-3B",
+    "checkpoint_revision": "a07cc9a04f16550a088caea529712d1d335b0ac1",
+    "continuous_batching": false,
+    "ep_degree": 1,
+    "fused_qkv": true,
+    "glu_mlp": true,
+    "local_ranks_size": 2,
+    "max_batch_size": 1,
+    "max_context_length": 1024,
+    "max_topk": 256,
+    "n_active_tokens": 1024,
+    "neuronxcc_version": "2.21.33363.0+82129205",
+    "on_device_sampling": true,
+    "optimum_neuron_version": "0.4.6.dev4",
+    "output_logits": false,
+    "pp_degree": 1,
+    "prefill_chunk_size": 0,
+    "sequence_length": 1024,
+    "speculation_length": 0,
+    "start_rank_id": 0,
+    "target": "trn1",
+    "torch_dtype": "bfloat16",
+    "tp_degree": 2
+  },
+  "no_rope_layer_interval": 4,
+  "no_rope_layers": [
+    1,
+    1,
+    1,
+    0,
+    1,
+    1,
+    1,
+    0,
+    1,
+    1,
+    1,
+    0,
+    1,
+    1,
+    1,
+    0,
+    1,
+    1,
+    1,
+    0,
+    1,
+    1,
+    1,
+    0,
+    1,
+    1,
+    1,
+    0,
+    1,
+    1,
+    1,
+    0,
+    1,
+    1,
+    1,
+    0
+  ],
+  "num_attention_heads": 16,
+  "num_hidden_layers": 36,
+  "num_key_value_heads": 4,
+  "pretraining_tp": 2,
+  "rms_norm_eps": 1e-06,
+  "rope_scaling": null,
+  "rope_theta": 5000000.0,
+  "sliding_window": null,
+  "use_cache": false,
+  "use_sliding_window": false,
+  "vocab_size": 128256
+}

neuronxcc-2.21.33363.0+82129205/MODULE_04c64527506049d877d9+b02446f6/compile_flags.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ ["--target=trn1", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "-O2", "--lnc=1", "--logfile=/tmp/nxd_vision_encoder/vision_encoder/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"]

neuronxcc-2.21.33363.0+82129205/MODULE_04c64527506049d877d9+b02446f6/model.done ADDED Viewed

File without changes

neuronxcc-2.21.33363.0+82129205/MODULE_04c64527506049d877d9+b02446f6/model.hlo_module.pb ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:b868910b39583cee3aeeab326d9fb6bfc1e12ce4e937d244abc110c48b64465e
+size 160725

neuronxcc-2.21.33363.0+82129205/MODULE_04c64527506049d877d9+b02446f6/model.neff ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:ae1f3f7cb2cc80137f4849385dad6bb0743f033cf7ede6d04c9dc4081d8e1d82
+size 30598144

neuronxcc-2.21.33363.0+82129205/MODULE_04c64527506049d877d9+b02446f6/wrapped_neff.hlo ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e4fc9bb782609dfaeb2b74d4fdc0108e917006c4f1622b3a05d2f0ebefc20b16
+size 30708309

neuronxcc-2.21.33363.0+82129205/MODULE_789b4718541dd2474859+a02c3a36/compile_flags.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ ["--target=trn1", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "-O2", "--lnc=1", "--logfile=/tmp/nxd_model/token_generation/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"]

neuronxcc-2.21.33363.0+82129205/MODULE_789b4718541dd2474859+a02c3a36/model.done ADDED Viewed

File without changes

neuronxcc-2.21.33363.0+82129205/MODULE_789b4718541dd2474859+a02c3a36/model.hlo_module.pb ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:2be0f520e77dfad494fc89a6f2bda248a6149afe1faaa16ff57adc5899bcefb2
+size 686960

neuronxcc-2.21.33363.0+82129205/MODULE_789b4718541dd2474859+a02c3a36/model.neff ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:422a22826c569c92260c9e11e4c7845725eb13de7f0943b0e6f012f3908c996b
+size 1485824

neuronxcc-2.21.33363.0+82129205/MODULE_789b4718541dd2474859+a02c3a36/wrapped_neff.hlo ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f412c2af4dc5c77c58211597836cecb60ff04a8741197af646a6c6b39fc137ab
+size 1596702

neuronxcc-2.21.33363.0+82129205/MODULE_82382ce0b34b3e272481+6170d8e1/compile_flags.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ ["--target=trn1", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "-O2", "--lnc=1", "--logfile=/tmp/nxd_model/chunked_prefill/_tp0_bk0/log-neuron-cc.txt"]

neuronxcc-2.21.33363.0+82129205/MODULE_82382ce0b34b3e272481+6170d8e1/model.done ADDED Viewed

File without changes

neuronxcc-2.21.33363.0+82129205/MODULE_82382ce0b34b3e272481+6170d8e1/model.hlo_module.pb ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:98739a3fd726ac3a8379af0d0b6e611702ebd66977e60a994ce9155f14e9cb3e
+size 911470

neuronxcc-2.21.33363.0+82129205/MODULE_82382ce0b34b3e272481+6170d8e1/model.neff ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:68e1150370b99c7ff1054699ec25421e64872fd7e32e9777e95cefc3c4e8e4a2
+size 11674624