diff --git a/.gitattributes b/.gitattributes index e370232a6bb3a2b53bde292cf4c7a797a111b040..dd5a0f9e0cad10312e12c390f3a250cdc5153173 100644 --- a/.gitattributes +++ b/.gitattributes @@ -7047,3 +7047,15 @@ neuronxcc-2.21.33363.0+82129205/MODULE_a36debd95d53c8bebd53+f7cce17f/model.neff neuronxcc-2.21.33363.0+82129205/MODULE_1b43d7c01692b5e7842e+9e8e849c/model.neff filter=lfs diff=lfs merge=lfs -text neuronxcc-2.21.33363.0+82129205/MODULE_95b9072f246645f24461+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text neuronxcc-2.21.33363.0+82129205/MODULE_95b9072f246645f24461+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text +neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text +neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text +neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/model.neff filter=lfs diff=lfs merge=lfs -text +neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text +neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text +neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/model.neff filter=lfs diff=lfs merge=lfs -text +neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/model.neff filter=lfs diff=lfs merge=lfs -text +neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/model.neff filter=lfs diff=lfs merge=lfs -text +neuronxcc-2.21.33363.0+82129205/MODULE_e6dd38f1c4ba59b167fc+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text +neuronxcc-2.21.33363.0+82129205/MODULE_e6dd38f1c4ba59b167fc+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text +neuronxcc-2.21.33363.0+82129205/MODULE_f8159368d9afab63b628+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text +neuronxcc-2.21.33363.0+82129205/MODULE_f8159368d9afab63b628+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text diff --git a/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/7890860ce2cf6931d044.json b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/7890860ce2cf6931d044.json new file mode 100644 index 0000000000000000000000000000000000000000..c1b96e43377074cb1165db10900591c561ec3d36 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/7890860ce2cf6931d044.json @@ -0,0 +1,62 @@ +{ + "_entry_class": "SingleModelCacheEntry", + "_model_id": "meta-llama/Llama-3.1-8B-Instruct", + "_task": "text-generation", + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "dtype": "bfloat16", + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 4096, + "initializer_range": 0.02, + "intermediate_size": 14336, + "max_position_embeddings": 131072, + "mlp_bias": false, + "model_type": "llama", + "neuron": { + "_serialized_key": "NxDNeuronConfig", + "batch_size": 32, + "capacity_factor": null, + "checkpoint_id": "meta-llama/Llama-3.1-8B-Instruct", + "checkpoint_revision": null, + "continuous_batching": false, + "ep_degree": 1, + "fused_qkv": true, + "glu_mlp": true, + "local_ranks_size": 8, + "max_batch_size": 32, + "max_context_length": 4096, + "max_topk": 256, + "n_active_tokens": 4096, + "neuronxcc_version": "2.21.33363.0+82129205", + "on_device_sampling": false, + "optimum_neuron_version": "0.4.5.dev2", + "output_logits": false, + "pp_degree": 1, + "sequence_length": 4096, + "speculation_length": 0, + "start_rank_id": 0, + "target": "trn1", + "torch_dtype": "bfloat16", + "tp_degree": 8 + }, + "num_attention_heads": 32, + "num_hidden_layers": 32, + "num_key_value_heads": 8, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_scaling": { + "factor": 8.0, + "high_freq_factor": 4.0, + "low_freq_factor": 1.0, + "original_max_position_embeddings": 8192, + "rope_type": "llama3" + }, + "rope_theta": 500000.0, + "tie_word_embeddings": false, + "use_cache": true, + "vocab_size": 128256 +} \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/bd4d1f3a2a63a9d40f94.json b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/bd4d1f3a2a63a9d40f94.json new file mode 100644 index 0000000000000000000000000000000000000000..6a95eddd43259c428176979bc0ff4338e3bbdebc --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/bd4d1f3a2a63a9d40f94.json @@ -0,0 +1,62 @@ +{ + "_entry_class": "SingleModelCacheEntry", + "_model_id": "meta-llama/Llama-3.1-8B-Instruct", + "_task": "text-generation", + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "dtype": "bfloat16", + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 4096, + "initializer_range": 0.02, + "intermediate_size": 14336, + "max_position_embeddings": 131072, + "mlp_bias": false, + "model_type": "llama", + "neuron": { + "_serialized_key": "NxDNeuronConfig", + "batch_size": 32, + "capacity_factor": null, + "checkpoint_id": "meta-llama/Llama-3.1-8B-Instruct", + "checkpoint_revision": null, + "continuous_batching": true, + "ep_degree": 1, + "fused_qkv": true, + "glu_mlp": true, + "local_ranks_size": 8, + "max_batch_size": 32, + "max_context_length": 4096, + "max_topk": 256, + "n_active_tokens": 4096, + "neuronxcc_version": "2.21.33363.0+82129205", + "on_device_sampling": false, + "optimum_neuron_version": "0.4.5.dev2", + "output_logits": false, + "pp_degree": 1, + "sequence_length": 4096, + "speculation_length": 0, + "start_rank_id": 0, + "target": "trn1", + "torch_dtype": "bfloat16", + "tp_degree": 8 + }, + "num_attention_heads": 32, + "num_hidden_layers": 32, + "num_key_value_heads": 8, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_scaling": { + "factor": 8.0, + "high_freq_factor": 4.0, + "low_freq_factor": 1.0, + "original_max_position_embeddings": 8192, + "rope_type": "llama3" + }, + "rope_theta": 500000.0, + "tie_word_embeddings": false, + "use_cache": true, + "vocab_size": 128256 +} \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/009cc4abe26e4522722e.json b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/009cc4abe26e4522722e.json new file mode 100644 index 0000000000000000000000000000000000000000..59fff19317ae7d2d5346c361e9c3df09a4a258e3 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/009cc4abe26e4522722e.json @@ -0,0 +1,63 @@ +{ + "_entry_class": "SingleModelCacheEntry", + "_model_id": "meta-llama/Llama-3.1-8B-Instruct", + "_task": "text-generation", + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "dtype": "bfloat16", + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 4096, + "initializer_range": 0.02, + "intermediate_size": 14336, + "max_position_embeddings": 131072, + "mlp_bias": false, + "model_type": "llama", + "neuron": { + "_serialized_key": "NxDNeuronConfig", + "batch_size": 32, + "capacity_factor": null, + "checkpoint_id": "meta-llama/Llama-3.1-8B-Instruct", + "checkpoint_revision": null, + "continuous_batching": true, + "ep_degree": 1, + "fused_qkv": true, + "glu_mlp": true, + "local_ranks_size": 8, + "max_batch_size": 32, + "max_context_length": 4096, + "max_topk": 256, + "n_active_tokens": 4096, + "neuronxcc_version": "2.21.33363.0+82129205", + "on_device_sampling": false, + "optimum_neuron_version": "0.4.6.dev1", + "output_logits": false, + "pp_degree": 1, + "prefill_chunk_size": 1024, + "sequence_length": 4096, + "speculation_length": 0, + "start_rank_id": 0, + "target": "trn1", + "torch_dtype": "bfloat16", + "tp_degree": 8 + }, + "num_attention_heads": 32, + "num_hidden_layers": 32, + "num_key_value_heads": 8, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_scaling": { + "factor": 8.0, + "high_freq_factor": 4.0, + "low_freq_factor": 1.0, + "original_max_position_embeddings": 8192, + "rope_type": "llama3" + }, + "rope_theta": 500000.0, + "tie_word_embeddings": false, + "use_cache": true, + "vocab_size": 128256 +} \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/449164b2be0b8d13730f.json b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/449164b2be0b8d13730f.json new file mode 100644 index 0000000000000000000000000000000000000000..392c7cc07089c367fc27ca6f46cc9584e32e00e1 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/449164b2be0b8d13730f.json @@ -0,0 +1,63 @@ +{ + "_entry_class": "SingleModelCacheEntry", + "_model_id": "meta-llama/Llama-3.1-8B-Instruct", + "_task": "text-generation", + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "dtype": "bfloat16", + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 4096, + "initializer_range": 0.02, + "intermediate_size": 14336, + "max_position_embeddings": 131072, + "mlp_bias": false, + "model_type": "llama", + "neuron": { + "_serialized_key": "NxDNeuronConfig", + "batch_size": 32, + "capacity_factor": null, + "checkpoint_id": "meta-llama/Llama-3.1-8B-Instruct", + "checkpoint_revision": null, + "continuous_batching": true, + "ep_degree": 1, + "fused_qkv": true, + "glu_mlp": true, + "local_ranks_size": 8, + "max_batch_size": 32, + "max_context_length": 4096, + "max_topk": 256, + "n_active_tokens": 4096, + "neuronxcc_version": "2.21.33363.0+82129205", + "on_device_sampling": true, + "optimum_neuron_version": "0.4.6.dev1", + "output_logits": false, + "pp_degree": 1, + "prefill_chunk_size": 0, + "sequence_length": 4096, + "speculation_length": 0, + "start_rank_id": 0, + "target": "trn1", + "torch_dtype": "bfloat16", + "tp_degree": 8 + }, + "num_attention_heads": 32, + "num_hidden_layers": 32, + "num_key_value_heads": 8, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_scaling": { + "factor": 8.0, + "high_freq_factor": 4.0, + "low_freq_factor": 1.0, + "original_max_position_embeddings": 8192, + "rope_type": "llama3" + }, + "rope_theta": 500000.0, + "tie_word_embeddings": false, + "use_cache": true, + "vocab_size": 128256 +} \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/5a9e3f4d03d538fb2f86.json b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/5a9e3f4d03d538fb2f86.json new file mode 100644 index 0000000000000000000000000000000000000000..111806b4a104b99a5e15bdf090b3c0ec22fdc772 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/5a9e3f4d03d538fb2f86.json @@ -0,0 +1,63 @@ +{ + "_entry_class": "SingleModelCacheEntry", + "_model_id": "meta-llama/Llama-3.1-8B-Instruct", + "_task": "text-generation", + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "dtype": "bfloat16", + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 4096, + "initializer_range": 0.02, + "intermediate_size": 14336, + "max_position_embeddings": 131072, + "mlp_bias": false, + "model_type": "llama", + "neuron": { + "_serialized_key": "NxDNeuronConfig", + "batch_size": 32, + "capacity_factor": null, + "checkpoint_id": "meta-llama/Llama-3.1-8B-Instruct", + "checkpoint_revision": null, + "continuous_batching": true, + "ep_degree": 1, + "fused_qkv": true, + "glu_mlp": true, + "local_ranks_size": 8, + "max_batch_size": 32, + "max_context_length": 4096, + "max_topk": 256, + "n_active_tokens": 4096, + "neuronxcc_version": "2.21.33363.0+82129205", + "on_device_sampling": false, + "optimum_neuron_version": "0.4.6.dev1", + "output_logits": false, + "pp_degree": 1, + "prefill_chunk_size": 0, + "sequence_length": 4096, + "speculation_length": 0, + "start_rank_id": 0, + "target": "trn1", + "torch_dtype": "bfloat16", + "tp_degree": 8 + }, + "num_attention_heads": 32, + "num_hidden_layers": 32, + "num_key_value_heads": 8, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_scaling": { + "factor": 8.0, + "high_freq_factor": 4.0, + "low_freq_factor": 1.0, + "original_max_position_embeddings": 8192, + "rope_type": "llama3" + }, + "rope_theta": 500000.0, + "tie_word_embeddings": false, + "use_cache": true, + "vocab_size": 128256 +} \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/7f7c6611f9515b012fbd.json b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/7f7c6611f9515b012fbd.json new file mode 100644 index 0000000000000000000000000000000000000000..f4656b20968226cf0d445d87ce4c8507557aa4e9 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/7f7c6611f9515b012fbd.json @@ -0,0 +1,63 @@ +{ + "_entry_class": "SingleModelCacheEntry", + "_model_id": "meta-llama/Llama-3.1-8B-Instruct", + "_task": "text-generation", + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "dtype": "bfloat16", + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 4096, + "initializer_range": 0.02, + "intermediate_size": 14336, + "max_position_embeddings": 131072, + "mlp_bias": false, + "model_type": "llama", + "neuron": { + "_serialized_key": "NxDNeuronConfig", + "batch_size": 32, + "capacity_factor": null, + "checkpoint_id": "meta-llama/Llama-3.1-8B-Instruct", + "checkpoint_revision": "0e9e39f249a16976918f6564b8830bc894c89659", + "continuous_batching": true, + "ep_degree": 1, + "fused_qkv": true, + "glu_mlp": true, + "local_ranks_size": 8, + "max_batch_size": 32, + "max_context_length": 4096, + "max_topk": 256, + "n_active_tokens": 4096, + "neuronxcc_version": "2.21.33363.0+82129205", + "on_device_sampling": true, + "optimum_neuron_version": "0.4.6.dev1", + "output_logits": false, + "pp_degree": 1, + "prefill_chunk_size": 1024, + "sequence_length": 4096, + "speculation_length": 0, + "start_rank_id": 0, + "target": "trn1", + "torch_dtype": "bfloat16", + "tp_degree": 8 + }, + "num_attention_heads": 32, + "num_hidden_layers": 32, + "num_key_value_heads": 8, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_scaling": { + "factor": 8.0, + "high_freq_factor": 4.0, + "low_freq_factor": 1.0, + "original_max_position_embeddings": 8192, + "rope_type": "llama3" + }, + "rope_theta": 500000.0, + "tie_word_embeddings": false, + "use_cache": true, + "vocab_size": 128256 +} \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/3b53548b380293de28b8.json b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/3b53548b380293de28b8.json new file mode 100644 index 0000000000000000000000000000000000000000..a3e0c668cd7d9144080eb107125dd361b18aacd7 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/3b53548b380293de28b8.json @@ -0,0 +1,64 @@ +{ + "_entry_class": "SingleModelCacheEntry", + "_model_id": "unsloth/Llama-3.2-1B-Instruct", + "_task": "text-generation", + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "dtype": "bfloat16", + "head_dim": 64, + "hidden_act": "silu", + "hidden_size": 2048, + "initializer_range": 0.02, + "intermediate_size": 8192, + "max_position_embeddings": 131072, + "mlp_bias": false, + "model_type": "llama", + "neuron": { + "_serialized_key": "NxDNeuronConfig", + "batch_size": 2, + "capacity_factor": null, + "checkpoint_id": "unsloth/Llama-3.2-1B-Instruct", + "checkpoint_revision": null, + "continuous_batching": false, + "ep_degree": 1, + "fused_qkv": true, + "glu_mlp": true, + "local_ranks_size": 2, + "max_batch_size": 2, + "max_context_length": 4096, + "max_topk": 256, + "n_active_tokens": 4096, + "neuronxcc_version": "2.21.33363.0+82129205", + "on_device_sampling": false, + "optimum_neuron_version": "0.4.6.dev2", + "output_logits": false, + "pp_degree": 1, + "prefill_chunk_size": 0, + "sequence_length": 4096, + "speculation_length": 0, + "start_rank_id": 0, + "target": "trn1", + "torch_dtype": "bfloat16", + "tp_degree": 2 + }, + "num_attention_heads": 32, + "num_hidden_layers": 16, + "num_key_value_heads": 8, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_scaling": { + "factor": 32.0, + "high_freq_factor": 4.0, + "low_freq_factor": 1.0, + "original_max_position_embeddings": 8192, + "rope_type": "llama3" + }, + "rope_theta": 500000.0, + "tie_word_embeddings": true, + "unsloth_fixed": true, + "use_cache": true, + "vocab_size": 128256 +} \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/700b41cac5912175b08c.json b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/700b41cac5912175b08c.json new file mode 100644 index 0000000000000000000000000000000000000000..7fcc97461422b0d598748ab3cf3b69ecbee2f726 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/700b41cac5912175b08c.json @@ -0,0 +1,64 @@ +{ + "_entry_class": "SingleModelCacheEntry", + "_model_id": "unsloth/Llama-3.2-1B-Instruct", + "_task": "text-generation", + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "dtype": "bfloat16", + "head_dim": 64, + "hidden_act": "silu", + "hidden_size": 2048, + "initializer_range": 0.02, + "intermediate_size": 8192, + "max_position_embeddings": 131072, + "mlp_bias": false, + "model_type": "llama", + "neuron": { + "_serialized_key": "NxDNeuronConfig", + "batch_size": 1, + "capacity_factor": null, + "checkpoint_id": "unsloth/Llama-3.2-1B-Instruct", + "checkpoint_revision": null, + "continuous_batching": false, + "ep_degree": 1, + "fused_qkv": true, + "glu_mlp": true, + "local_ranks_size": 2, + "max_batch_size": 1, + "max_context_length": 512, + "max_topk": 256, + "n_active_tokens": 512, + "neuronxcc_version": "2.21.33363.0+82129205", + "on_device_sampling": false, + "optimum_neuron_version": "0.4.6.dev2", + "output_logits": false, + "pp_degree": 1, + "prefill_chunk_size": 0, + "sequence_length": 512, + "speculation_length": 0, + "start_rank_id": 0, + "target": "trn1", + "torch_dtype": "bfloat16", + "tp_degree": 2 + }, + "num_attention_heads": 32, + "num_hidden_layers": 16, + "num_key_value_heads": 8, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_scaling": { + "factor": 32.0, + "high_freq_factor": 4.0, + "low_freq_factor": 1.0, + "original_max_position_embeddings": 8192, + "rope_type": "llama3" + }, + "rope_theta": 500000.0, + "tie_word_embeddings": true, + "unsloth_fixed": true, + "use_cache": true, + "vocab_size": 128256 +} \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/792c553a46127d9f996b.json b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/792c553a46127d9f996b.json new file mode 100644 index 0000000000000000000000000000000000000000..44c3d83b2c400aa860ee99124ba160c9f94641c9 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/792c553a46127d9f996b.json @@ -0,0 +1,64 @@ +{ + "_entry_class": "SingleModelCacheEntry", + "_model_id": "unsloth/Llama-3.2-1B-Instruct", + "_task": "text-generation", + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "dtype": "bfloat16", + "head_dim": 64, + "hidden_act": "silu", + "hidden_size": 2048, + "initializer_range": 0.02, + "intermediate_size": 8192, + "max_position_embeddings": 131072, + "mlp_bias": false, + "model_type": "llama", + "neuron": { + "_serialized_key": "NxDNeuronConfig", + "batch_size": 1, + "capacity_factor": null, + "checkpoint_id": "unsloth/Llama-3.2-1B-Instruct", + "checkpoint_revision": null, + "continuous_batching": false, + "ep_degree": 1, + "fused_qkv": true, + "glu_mlp": true, + "local_ranks_size": 2, + "max_batch_size": 1, + "max_context_length": 512, + "max_topk": 256, + "n_active_tokens": 512, + "neuronxcc_version": "2.21.33363.0+82129205", + "on_device_sampling": false, + "optimum_neuron_version": "0.4.6.dev2", + "output_logits": false, + "pp_degree": 1, + "prefill_chunk_size": 64, + "sequence_length": 512, + "speculation_length": 0, + "start_rank_id": 0, + "target": "trn1", + "torch_dtype": "bfloat16", + "tp_degree": 2 + }, + "num_attention_heads": 32, + "num_hidden_layers": 16, + "num_key_value_heads": 8, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_scaling": { + "factor": 32.0, + "high_freq_factor": 4.0, + "low_freq_factor": 1.0, + "original_max_position_embeddings": 8192, + "rope_type": "llama3" + }, + "rope_theta": 500000.0, + "tie_word_embeddings": true, + "unsloth_fixed": true, + "use_cache": true, + "vocab_size": 128256 +} \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/llama/unsloth/Llama-3.2-1B-Instruct/700b41cac5912175b08c.json b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/llama/unsloth/Llama-3.2-1B-Instruct/700b41cac5912175b08c.json new file mode 100644 index 0000000000000000000000000000000000000000..7fcc97461422b0d598748ab3cf3b69ecbee2f726 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/llama/unsloth/Llama-3.2-1B-Instruct/700b41cac5912175b08c.json @@ -0,0 +1,64 @@ +{ + "_entry_class": "SingleModelCacheEntry", + "_model_id": "unsloth/Llama-3.2-1B-Instruct", + "_task": "text-generation", + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "dtype": "bfloat16", + "head_dim": 64, + "hidden_act": "silu", + "hidden_size": 2048, + "initializer_range": 0.02, + "intermediate_size": 8192, + "max_position_embeddings": 131072, + "mlp_bias": false, + "model_type": "llama", + "neuron": { + "_serialized_key": "NxDNeuronConfig", + "batch_size": 1, + "capacity_factor": null, + "checkpoint_id": "unsloth/Llama-3.2-1B-Instruct", + "checkpoint_revision": null, + "continuous_batching": false, + "ep_degree": 1, + "fused_qkv": true, + "glu_mlp": true, + "local_ranks_size": 2, + "max_batch_size": 1, + "max_context_length": 512, + "max_topk": 256, + "n_active_tokens": 512, + "neuronxcc_version": "2.21.33363.0+82129205", + "on_device_sampling": false, + "optimum_neuron_version": "0.4.6.dev2", + "output_logits": false, + "pp_degree": 1, + "prefill_chunk_size": 0, + "sequence_length": 512, + "speculation_length": 0, + "start_rank_id": 0, + "target": "trn1", + "torch_dtype": "bfloat16", + "tp_degree": 2 + }, + "num_attention_heads": 32, + "num_hidden_layers": 16, + "num_key_value_heads": 8, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_scaling": { + "factor": 32.0, + "high_freq_factor": 4.0, + "low_freq_factor": 1.0, + "original_max_position_embeddings": 8192, + "rope_type": "llama3" + }, + "rope_theta": 500000.0, + "tie_word_embeddings": true, + "unsloth_fixed": true, + "use_cache": true, + "vocab_size": 128256 +} \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_15104978417860996248+e30acd3a/model.neff b/neuronxcc-2.21.33363.0+82129205/MODULE_15104978417860996248+e30acd3a/model.neff index cb2da8c4eace83c30ce7aab0077751bab7a157c3..d43d6aa659e9d6766f27e0481ee0bc38c9bdb80a 100644 Binary files a/neuronxcc-2.21.33363.0+82129205/MODULE_15104978417860996248+e30acd3a/model.neff and b/neuronxcc-2.21.33363.0+82129205/MODULE_15104978417860996248+e30acd3a/model.neff differ diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/compile_flags.json b/neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/compile_flags.json new file mode 100644 index 0000000000000000000000000000000000000000..5726abc7d1d8c52fa95bc7919439a23a23fe3b9a --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/compile_flags.json @@ -0,0 +1 @@ +["--target=trn1", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "-O2", "--lnc=1", "--logfile=/tmp/nxd_model/token_generation/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"] \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/model.done b/neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/model.done new file mode 100644 index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/model.hlo_module.pb b/neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/model.hlo_module.pb new file mode 100644 index 0000000000000000000000000000000000000000..c9bbef884b54ea37a322bd215f5cd1c5730f5725 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/model.hlo_module.pb @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fc384fc0324b11db5da96604331cd5055d6c75432751e5e40d6a8365b8fa648a +size 799960 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/model.neff b/neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/model.neff new file mode 100644 index 0000000000000000000000000000000000000000..0610c39bc68f33f657392dc88d3288d23c41febd --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/model.neff @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a40ec25bcacdb7caad8ec72238d804a702ac2b36d50f2eae5452615372333e55 +size 6872064 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/wrapped_neff.hlo b/neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/wrapped_neff.hlo new file mode 100644 index 0000000000000000000000000000000000000000..65f729a0a4c812461a01ec1eb63ddcaa6b617a37 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/wrapped_neff.hlo @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ccb46abb3a38777c337bac220a060af8fe90ec103f8eb1d8e99a288592f166d6 +size 7019528 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/compile_flags.json b/neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/compile_flags.json new file mode 100644 index 0000000000000000000000000000000000000000..9c277888420f00defd99fc3c102007a98b09199d --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/compile_flags.json @@ -0,0 +1 @@ +["--target=trn1", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "-O2", "--lnc=1", "--logfile=/tmp/nxd_model/chunked_prefill/_tp0_bk0/log-neuron-cc.txt"] \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/model.done b/neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/model.done new file mode 100644 index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/model.hlo_module.pb b/neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/model.hlo_module.pb new file mode 100644 index 0000000000000000000000000000000000000000..c0907c6b728a0842d24a57b61caa1e024f47b4a8 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/model.hlo_module.pb @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f26bf8ee45db3f0f1962d5413d680ac63ccb89bc66ab6282e0e02415fa8f5e2c +size 498827 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/model.neff b/neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/model.neff new file mode 100644 index 0000000000000000000000000000000000000000..e9a9fc11873173fc6b642258ffe2a2a910dcc781 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/model.neff @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:674273a150bfda8e7977bae1985e99a62ef564db8572385d157240ffe7600813 +size 2120704 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/compile_flags.json b/neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/compile_flags.json new file mode 100644 index 0000000000000000000000000000000000000000..5726abc7d1d8c52fa95bc7919439a23a23fe3b9a --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/compile_flags.json @@ -0,0 +1 @@ +["--target=trn1", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "-O2", "--lnc=1", "--logfile=/tmp/nxd_model/token_generation/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"] \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/model.done b/neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/model.done new file mode 100644 index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/model.hlo_module.pb b/neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/model.hlo_module.pb new file mode 100644 index 0000000000000000000000000000000000000000..1a9af0c0ad6f5643121c27682c5561d12a42b950 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/model.hlo_module.pb @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:206450a370d213a0ad9a7eb0ddfd4050292bf64ecf17344b02204c661dcfabf3 +size 397298 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/model.neff b/neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/model.neff new file mode 100644 index 0000000000000000000000000000000000000000..c731837aaf7916548e06cb34ade3e571c1b7a3fa --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/model.neff @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1aae8abec5aee4c1fcae30792817224d16127cd62c27bec469cd8f9802bd1dbd +size 1741824 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/wrapped_neff.hlo b/neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/wrapped_neff.hlo new file mode 100644 index 0000000000000000000000000000000000000000..90ec627a25ba63328ed395671b00596c054e7af0 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/wrapped_neff.hlo @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4852707561e1f28060ba885c808186aeecdf0c1389b0417de838d09cce390175 +size 1815793 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/compile_flags.json b/neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/compile_flags.json new file mode 100644 index 0000000000000000000000000000000000000000..ebc1946b1a734cf0cae3ce1488a867b440bccf29 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/compile_flags.json @@ -0,0 +1 @@ +["--target=trn1", "--enable-saturate-infinity", "--auto-cast=none", "--model-type=transformer", "-O1", "--logfile=/tmp/nxdi_test_48ffe586-34f9-4592-bf76-89fbf67ad5e6/compiler_workdir/SoftmaxNoMask/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"] \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/model.done b/neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/model.done new file mode 100644 index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/model.hlo_module.pb b/neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/model.hlo_module.pb new file mode 100644 index 0000000000000000000000000000000000000000..36897c006be58fcc135059831fe41ce96ad676a1 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/model.hlo_module.pb @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8e542c76cf27c3f807feac47dd3f58fd4c308e8d9e932f0732774bee19d640d8 +size 3881 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/model.neff b/neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/model.neff new file mode 100644 index 0000000000000000000000000000000000000000..cab330cfc8ae20de4207c39deb3df9c43ef32881 Binary files /dev/null and b/neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/model.neff differ diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/compile_flags.json b/neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/compile_flags.json new file mode 100644 index 0000000000000000000000000000000000000000..836724f44545ce0dedda1521fd4c623a6ea8ec72 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/compile_flags.json @@ -0,0 +1 @@ +["--target=trn1", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "-O2", "--lnc=1", "--logfile=/tmp/nxd_model/context_encoding/_tp0_bk0/log-neuron-cc.txt"] \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/model.done b/neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/model.done new file mode 100644 index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/model.hlo_module.pb b/neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/model.hlo_module.pb new file mode 100644 index 0000000000000000000000000000000000000000..b499c25c5dbf5e108553fcd9890fcaf15063dfd9 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/model.hlo_module.pb @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a945df4dd154fab077f1da24bde6fb2e46aa20de51c6cc39d5545ffb7c64d733 +size 337649 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/model.neff b/neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/model.neff new file mode 100644 index 0000000000000000000000000000000000000000..8b9bb88e10d03139a5df995ae507cd276087bbd3 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/model.neff @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0bcae3ba218cc15ae1709d588223e5a1c5e2a512a6d5156146c0cbafdec18162 +size 2612224 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/compile_flags.json b/neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/compile_flags.json new file mode 100644 index 0000000000000000000000000000000000000000..66dbbc380b38eab1cda67a61e8d1faa8662e043f --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/compile_flags.json @@ -0,0 +1 @@ +["--target=trn1", "--enable-saturate-infinity", "--auto-cast=none", "--model-type=transformer", "-O1", "--logfile=/tmp/nxdi_test_75ae5c33-5e1e-46c6-8477-19333053591f/compiler_workdir/SoftmaxNoMask/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"] \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/model.done b/neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/model.done new file mode 100644 index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/model.hlo_module.pb b/neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/model.hlo_module.pb new file mode 100644 index 0000000000000000000000000000000000000000..c213570f630857303e7b0ec653018546c02dc983 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/model.hlo_module.pb @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5efb0ea01dc0241314f91e19cbd9be551d1a66e7dbadd7cd4d6473d6b7d4b99a +size 3881 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/model.neff b/neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/model.neff new file mode 100644 index 0000000000000000000000000000000000000000..a7909ef9fba2c348d8624cc9a77646676204a945 Binary files /dev/null and b/neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/model.neff differ diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/compile_flags.json b/neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/compile_flags.json new file mode 100644 index 0000000000000000000000000000000000000000..9c277888420f00defd99fc3c102007a98b09199d --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/compile_flags.json @@ -0,0 +1 @@ +["--target=trn1", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "-O2", "--lnc=1", "--logfile=/tmp/nxd_model/chunked_prefill/_tp0_bk0/log-neuron-cc.txt"] \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/model.done b/neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/model.done new file mode 100644 index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/model.hlo_module.pb b/neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/model.hlo_module.pb new file mode 100644 index 0000000000000000000000000000000000000000..a8ce0ac3129de5d9dbb2e4b8472860afa14782fc --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/model.hlo_module.pb @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3296a30cf85cdc9498292d7213649898ef13ed89a3910ca123d7526eb9824e97 +size 1034416 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/model.neff b/neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/model.neff new file mode 100644 index 0000000000000000000000000000000000000000..877a14dae3281b9d102cd3ad708744d9fdafeea7 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/model.neff @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6fa161991c7b70c2283cfd5351df193872e23bc89b0d68deca4e6fc5a6b74bb9 +size 18996224 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/compile_flags.json b/neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/compile_flags.json new file mode 100644 index 0000000000000000000000000000000000000000..a5bdc12875b0657f1f1172e98b31b3f17f081594 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/compile_flags.json @@ -0,0 +1 @@ +["--target=trn1", "--enable-saturate-infinity", "--auto-cast=none", "--model-type=transformer", "-O1", "--logfile=/tmp/nxdi_test_bde15d7b-1637-4861-8418-410426f6fc99/compiler_workdir/SoftmaxWithMask/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"] \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/model.done b/neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/model.done new file mode 100644 index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/model.hlo_module.pb b/neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/model.hlo_module.pb new file mode 100644 index 0000000000000000000000000000000000000000..f2ee42eb9f1aa790cfddd8e066da7c8aeb1214d9 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/model.hlo_module.pb @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:70f96487dad4bb02b98bf2c955fe59650a5fdbcf1d763fdf56ec412b62b5774c +size 5596 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/model.neff b/neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/model.neff new file mode 100644 index 0000000000000000000000000000000000000000..b7d002d2db2364f07aa6a5ac83519e2342c524fd Binary files /dev/null and b/neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/model.neff differ diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/compile_flags.json b/neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/compile_flags.json new file mode 100644 index 0000000000000000000000000000000000000000..836724f44545ce0dedda1521fd4c623a6ea8ec72 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/compile_flags.json @@ -0,0 +1 @@ +["--target=trn1", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "-O2", "--lnc=1", "--logfile=/tmp/nxd_model/context_encoding/_tp0_bk0/log-neuron-cc.txt"] \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/model.done b/neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/model.done new file mode 100644 index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/model.hlo_module.pb b/neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/model.hlo_module.pb new file mode 100644 index 0000000000000000000000000000000000000000..87dbb55010b1d1534d81b1b54e30e8e126ee2a82 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/model.hlo_module.pb @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8eb71fdd81d66f5a362985d7ac7802aa3739bc1d902de7ecff7b933bf65bfcec +size 861207 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/model.neff b/neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/model.neff new file mode 100644 index 0000000000000000000000000000000000000000..1a7a13ba14530332360c2e25f7544a23bd24cf48 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/model.neff @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fdff315f0a385fe46b91a0fc08f91f35ddd537601a850735ca27314c468efc9d +size 60457984 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_bbb38ecd8be44b576b6c+d0e5475d/compile_flags.json b/neuronxcc-2.21.33363.0+82129205/MODULE_bbb38ecd8be44b576b6c+d0e5475d/compile_flags.json new file mode 100644 index 0000000000000000000000000000000000000000..46ab8ff4e937108bfb628f7fdb6299ab4fd7f091 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_bbb38ecd8be44b576b6c+d0e5475d/compile_flags.json @@ -0,0 +1 @@ +["--target=trn1", "--enable-saturate-infinity", "--auto-cast=none", "--model-type=transformer", "-O1", "--logfile=/tmp/nxdi_test_a0d71bcb-0d68-4d9b-8be4-e3e2015e983b/compiler_workdir/SoftmaxWithMask/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"] \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_bbb38ecd8be44b576b6c+d0e5475d/model.done b/neuronxcc-2.21.33363.0+82129205/MODULE_bbb38ecd8be44b576b6c+d0e5475d/model.done new file mode 100644 index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_bbb38ecd8be44b576b6c+d0e5475d/model.hlo_module.pb b/neuronxcc-2.21.33363.0+82129205/MODULE_bbb38ecd8be44b576b6c+d0e5475d/model.hlo_module.pb new file mode 100644 index 0000000000000000000000000000000000000000..1122d44653d734af6a44cd149690f70cdefd9b05 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_bbb38ecd8be44b576b6c+d0e5475d/model.hlo_module.pb @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f9a258ce111db27dc5ba46ec4f9f6877c56b0777a5534c0eba14488ff9eaf298 +size 5596 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_bbb38ecd8be44b576b6c+d0e5475d/model.neff b/neuronxcc-2.21.33363.0+82129205/MODULE_bbb38ecd8be44b576b6c+d0e5475d/model.neff new file mode 100644 index 0000000000000000000000000000000000000000..e106300bc38edafd2048c71eaf0e1f76a206dced Binary files /dev/null and b/neuronxcc-2.21.33363.0+82129205/MODULE_bbb38ecd8be44b576b6c+d0e5475d/model.neff differ diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_e6dd38f1c4ba59b167fc+a02c3a36/compile_flags.json b/neuronxcc-2.21.33363.0+82129205/MODULE_e6dd38f1c4ba59b167fc+a02c3a36/compile_flags.json new file mode 100644 index 0000000000000000000000000000000000000000..5726abc7d1d8c52fa95bc7919439a23a23fe3b9a --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_e6dd38f1c4ba59b167fc+a02c3a36/compile_flags.json @@ -0,0 +1 @@ +["--target=trn1", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "-O2", "--lnc=1", "--logfile=/tmp/nxd_model/token_generation/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"] \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_e6dd38f1c4ba59b167fc+a02c3a36/model.done b/neuronxcc-2.21.33363.0+82129205/MODULE_e6dd38f1c4ba59b167fc+a02c3a36/model.done new file mode 100644 index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_e6dd38f1c4ba59b167fc+a02c3a36/model.hlo_module.pb b/neuronxcc-2.21.33363.0+82129205/MODULE_e6dd38f1c4ba59b167fc+a02c3a36/model.hlo_module.pb new file mode 100644 index 0000000000000000000000000000000000000000..774b8065eeabb7736832ac7e09ddfb069f41b0e1 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_e6dd38f1c4ba59b167fc+a02c3a36/model.hlo_module.pb @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e92d2d0d20ae978067e84f92ce9ec0d7846dd4d65d0aa94885cb65aca7430283 +size 756502 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_e6dd38f1c4ba59b167fc+a02c3a36/model.neff b/neuronxcc-2.21.33363.0+82129205/MODULE_e6dd38f1c4ba59b167fc+a02c3a36/model.neff new file mode 100644 index 0000000000000000000000000000000000000000..717de408e295b32eeb2f4a09a6f79c90832470f3 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_e6dd38f1c4ba59b167fc+a02c3a36/model.neff @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:201e5fcb9e03db681b82af2e4e7b68aa6de2b9340609d5788419fd7a3c762d90 +size 6902784 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_e6dd38f1c4ba59b167fc+a02c3a36/wrapped_neff.hlo b/neuronxcc-2.21.33363.0+82129205/MODULE_e6dd38f1c4ba59b167fc+a02c3a36/wrapped_neff.hlo new file mode 100644 index 0000000000000000000000000000000000000000..3b3176295603f39510dbf6fc3478f13329d17cd6 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_e6dd38f1c4ba59b167fc+a02c3a36/wrapped_neff.hlo @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fa4c57f0ea1fc1a29171438254b78e93b3dafa6b7ca76d7f239821636febb55b +size 7050133 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_f8159368d9afab63b628+a02c3a36/compile_flags.json b/neuronxcc-2.21.33363.0+82129205/MODULE_f8159368d9afab63b628+a02c3a36/compile_flags.json new file mode 100644 index 0000000000000000000000000000000000000000..5726abc7d1d8c52fa95bc7919439a23a23fe3b9a --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_f8159368d9afab63b628+a02c3a36/compile_flags.json @@ -0,0 +1 @@ +["--target=trn1", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "-O2", "--lnc=1", "--logfile=/tmp/nxd_model/token_generation/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"] \ No newline at end of file diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_f8159368d9afab63b628+a02c3a36/model.done b/neuronxcc-2.21.33363.0+82129205/MODULE_f8159368d9afab63b628+a02c3a36/model.done new file mode 100644 index 0000000000000000000000000000000000000000..e69de29bb2d1d6434b8b29ae775ad8c2e48c5391 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_f8159368d9afab63b628+a02c3a36/model.hlo_module.pb b/neuronxcc-2.21.33363.0+82129205/MODULE_f8159368d9afab63b628+a02c3a36/model.hlo_module.pb new file mode 100644 index 0000000000000000000000000000000000000000..8911557c3a6c337b6adc4fa2a04abe66b38e6f08 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_f8159368d9afab63b628+a02c3a36/model.hlo_module.pb @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:74ad8e0da8985a27c91ac3509ec8904e764954ce5d228803b0557bb14e03cb5f +size 404935 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_f8159368d9afab63b628+a02c3a36/model.neff b/neuronxcc-2.21.33363.0+82129205/MODULE_f8159368d9afab63b628+a02c3a36/model.neff new file mode 100644 index 0000000000000000000000000000000000000000..80fae216dbee869b7a44b83233e83b0ca4734f2c --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_f8159368d9afab63b628+a02c3a36/model.neff @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4a45354a1b108e39b68b6f156c167ffc571926e059231995ce6c4b754614e35f +size 2479104 diff --git a/neuronxcc-2.21.33363.0+82129205/MODULE_f8159368d9afab63b628+a02c3a36/wrapped_neff.hlo b/neuronxcc-2.21.33363.0+82129205/MODULE_f8159368d9afab63b628+a02c3a36/wrapped_neff.hlo new file mode 100644 index 0000000000000000000000000000000000000000..b9bed061e71b2a1a6b69785b4c62aa1d6985d2f1 --- /dev/null +++ b/neuronxcc-2.21.33363.0+82129205/MODULE_f8159368d9afab63b628+a02c3a36/wrapped_neff.hlo @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9c1404efe9f63a5c6e589a8cfcb8378a7dbc36c42a4d3a47696f8fc893dc6e43 +size 2553076