dacorvo HF Staff commited on
Commit
95799b9
·
verified ·
1 Parent(s): 2f8c0b8

Synchronizing local compiler cache.

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +12 -0
  2. neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/7890860ce2cf6931d044.json +62 -0
  3. neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/bd4d1f3a2a63a9d40f94.json +62 -0
  4. neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/009cc4abe26e4522722e.json +63 -0
  5. neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/449164b2be0b8d13730f.json +63 -0
  6. neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/5a9e3f4d03d538fb2f86.json +63 -0
  7. neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/7f7c6611f9515b012fbd.json +63 -0
  8. neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/3b53548b380293de28b8.json +64 -0
  9. neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/700b41cac5912175b08c.json +64 -0
  10. neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/792c553a46127d9f996b.json +64 -0
  11. neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/llama/unsloth/Llama-3.2-1B-Instruct/700b41cac5912175b08c.json +64 -0
  12. neuronxcc-2.21.33363.0+82129205/MODULE_15104978417860996248+e30acd3a/model.neff +0 -0
  13. neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/compile_flags.json +1 -0
  14. neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/model.done +0 -0
  15. neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/model.hlo_module.pb +3 -0
  16. neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/model.neff +3 -0
  17. neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/wrapped_neff.hlo +3 -0
  18. neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/compile_flags.json +1 -0
  19. neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/model.done +0 -0
  20. neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/model.hlo_module.pb +3 -0
  21. neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/model.neff +3 -0
  22. neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/compile_flags.json +1 -0
  23. neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/model.done +0 -0
  24. neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/model.hlo_module.pb +3 -0
  25. neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/model.neff +3 -0
  26. neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/wrapped_neff.hlo +3 -0
  27. neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/compile_flags.json +1 -0
  28. neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/model.done +0 -0
  29. neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/model.hlo_module.pb +3 -0
  30. neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/model.neff +0 -0
  31. neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/compile_flags.json +1 -0
  32. neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/model.done +0 -0
  33. neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/model.hlo_module.pb +3 -0
  34. neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/model.neff +3 -0
  35. neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/compile_flags.json +1 -0
  36. neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/model.done +0 -0
  37. neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/model.hlo_module.pb +3 -0
  38. neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/model.neff +0 -0
  39. neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/compile_flags.json +1 -0
  40. neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/model.done +0 -0
  41. neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/model.hlo_module.pb +3 -0
  42. neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/model.neff +3 -0
  43. neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/compile_flags.json +1 -0
  44. neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/model.done +0 -0
  45. neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/model.hlo_module.pb +3 -0
  46. neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/model.neff +0 -0
  47. neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/compile_flags.json +1 -0
  48. neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/model.done +0 -0
  49. neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/model.hlo_module.pb +3 -0
  50. neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/model.neff +3 -0
.gitattributes CHANGED
@@ -7047,3 +7047,15 @@ neuronxcc-2.21.33363.0+82129205/MODULE_a36debd95d53c8bebd53+f7cce17f/model.neff
7047
  neuronxcc-2.21.33363.0+82129205/MODULE_1b43d7c01692b5e7842e+9e8e849c/model.neff filter=lfs diff=lfs merge=lfs -text
7048
  neuronxcc-2.21.33363.0+82129205/MODULE_95b9072f246645f24461+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
7049
  neuronxcc-2.21.33363.0+82129205/MODULE_95b9072f246645f24461+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
7047
  neuronxcc-2.21.33363.0+82129205/MODULE_1b43d7c01692b5e7842e+9e8e849c/model.neff filter=lfs diff=lfs merge=lfs -text
7048
  neuronxcc-2.21.33363.0+82129205/MODULE_95b9072f246645f24461+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
7049
  neuronxcc-2.21.33363.0+82129205/MODULE_95b9072f246645f24461+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
7050
+ neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
7051
+ neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
7052
+ neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/model.neff filter=lfs diff=lfs merge=lfs -text
7053
+ neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
7054
+ neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
7055
+ neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
7056
+ neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/model.neff filter=lfs diff=lfs merge=lfs -text
7057
+ neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/model.neff filter=lfs diff=lfs merge=lfs -text
7058
+ neuronxcc-2.21.33363.0+82129205/MODULE_e6dd38f1c4ba59b167fc+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
7059
+ neuronxcc-2.21.33363.0+82129205/MODULE_e6dd38f1c4ba59b167fc+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
7060
+ neuronxcc-2.21.33363.0+82129205/MODULE_f8159368d9afab63b628+a02c3a36/model.neff filter=lfs diff=lfs merge=lfs -text
7061
+ neuronxcc-2.21.33363.0+82129205/MODULE_f8159368d9afab63b628+a02c3a36/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/7890860ce2cf6931d044.json ADDED
@@ -0,0 +1,62 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_entry_class": "SingleModelCacheEntry",
3
+ "_model_id": "meta-llama/Llama-3.1-8B-Instruct",
4
+ "_task": "text-generation",
5
+ "architectures": [
6
+ "LlamaForCausalLM"
7
+ ],
8
+ "attention_bias": false,
9
+ "attention_dropout": 0.0,
10
+ "dtype": "bfloat16",
11
+ "head_dim": 128,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 4096,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 14336,
16
+ "max_position_embeddings": 131072,
17
+ "mlp_bias": false,
18
+ "model_type": "llama",
19
+ "neuron": {
20
+ "_serialized_key": "NxDNeuronConfig",
21
+ "batch_size": 32,
22
+ "capacity_factor": null,
23
+ "checkpoint_id": "meta-llama/Llama-3.1-8B-Instruct",
24
+ "checkpoint_revision": null,
25
+ "continuous_batching": false,
26
+ "ep_degree": 1,
27
+ "fused_qkv": true,
28
+ "glu_mlp": true,
29
+ "local_ranks_size": 8,
30
+ "max_batch_size": 32,
31
+ "max_context_length": 4096,
32
+ "max_topk": 256,
33
+ "n_active_tokens": 4096,
34
+ "neuronxcc_version": "2.21.33363.0+82129205",
35
+ "on_device_sampling": false,
36
+ "optimum_neuron_version": "0.4.5.dev2",
37
+ "output_logits": false,
38
+ "pp_degree": 1,
39
+ "sequence_length": 4096,
40
+ "speculation_length": 0,
41
+ "start_rank_id": 0,
42
+ "target": "trn1",
43
+ "torch_dtype": "bfloat16",
44
+ "tp_degree": 8
45
+ },
46
+ "num_attention_heads": 32,
47
+ "num_hidden_layers": 32,
48
+ "num_key_value_heads": 8,
49
+ "pretraining_tp": 1,
50
+ "rms_norm_eps": 1e-05,
51
+ "rope_scaling": {
52
+ "factor": 8.0,
53
+ "high_freq_factor": 4.0,
54
+ "low_freq_factor": 1.0,
55
+ "original_max_position_embeddings": 8192,
56
+ "rope_type": "llama3"
57
+ },
58
+ "rope_theta": 500000.0,
59
+ "tie_word_embeddings": false,
60
+ "use_cache": true,
61
+ "vocab_size": 128256
62
+ }
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.5.dev2/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/bd4d1f3a2a63a9d40f94.json ADDED
@@ -0,0 +1,62 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_entry_class": "SingleModelCacheEntry",
3
+ "_model_id": "meta-llama/Llama-3.1-8B-Instruct",
4
+ "_task": "text-generation",
5
+ "architectures": [
6
+ "LlamaForCausalLM"
7
+ ],
8
+ "attention_bias": false,
9
+ "attention_dropout": 0.0,
10
+ "dtype": "bfloat16",
11
+ "head_dim": 128,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 4096,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 14336,
16
+ "max_position_embeddings": 131072,
17
+ "mlp_bias": false,
18
+ "model_type": "llama",
19
+ "neuron": {
20
+ "_serialized_key": "NxDNeuronConfig",
21
+ "batch_size": 32,
22
+ "capacity_factor": null,
23
+ "checkpoint_id": "meta-llama/Llama-3.1-8B-Instruct",
24
+ "checkpoint_revision": null,
25
+ "continuous_batching": true,
26
+ "ep_degree": 1,
27
+ "fused_qkv": true,
28
+ "glu_mlp": true,
29
+ "local_ranks_size": 8,
30
+ "max_batch_size": 32,
31
+ "max_context_length": 4096,
32
+ "max_topk": 256,
33
+ "n_active_tokens": 4096,
34
+ "neuronxcc_version": "2.21.33363.0+82129205",
35
+ "on_device_sampling": false,
36
+ "optimum_neuron_version": "0.4.5.dev2",
37
+ "output_logits": false,
38
+ "pp_degree": 1,
39
+ "sequence_length": 4096,
40
+ "speculation_length": 0,
41
+ "start_rank_id": 0,
42
+ "target": "trn1",
43
+ "torch_dtype": "bfloat16",
44
+ "tp_degree": 8
45
+ },
46
+ "num_attention_heads": 32,
47
+ "num_hidden_layers": 32,
48
+ "num_key_value_heads": 8,
49
+ "pretraining_tp": 1,
50
+ "rms_norm_eps": 1e-05,
51
+ "rope_scaling": {
52
+ "factor": 8.0,
53
+ "high_freq_factor": 4.0,
54
+ "low_freq_factor": 1.0,
55
+ "original_max_position_embeddings": 8192,
56
+ "rope_type": "llama3"
57
+ },
58
+ "rope_theta": 500000.0,
59
+ "tie_word_embeddings": false,
60
+ "use_cache": true,
61
+ "vocab_size": 128256
62
+ }
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/009cc4abe26e4522722e.json ADDED
@@ -0,0 +1,63 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_entry_class": "SingleModelCacheEntry",
3
+ "_model_id": "meta-llama/Llama-3.1-8B-Instruct",
4
+ "_task": "text-generation",
5
+ "architectures": [
6
+ "LlamaForCausalLM"
7
+ ],
8
+ "attention_bias": false,
9
+ "attention_dropout": 0.0,
10
+ "dtype": "bfloat16",
11
+ "head_dim": 128,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 4096,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 14336,
16
+ "max_position_embeddings": 131072,
17
+ "mlp_bias": false,
18
+ "model_type": "llama",
19
+ "neuron": {
20
+ "_serialized_key": "NxDNeuronConfig",
21
+ "batch_size": 32,
22
+ "capacity_factor": null,
23
+ "checkpoint_id": "meta-llama/Llama-3.1-8B-Instruct",
24
+ "checkpoint_revision": null,
25
+ "continuous_batching": true,
26
+ "ep_degree": 1,
27
+ "fused_qkv": true,
28
+ "glu_mlp": true,
29
+ "local_ranks_size": 8,
30
+ "max_batch_size": 32,
31
+ "max_context_length": 4096,
32
+ "max_topk": 256,
33
+ "n_active_tokens": 4096,
34
+ "neuronxcc_version": "2.21.33363.0+82129205",
35
+ "on_device_sampling": false,
36
+ "optimum_neuron_version": "0.4.6.dev1",
37
+ "output_logits": false,
38
+ "pp_degree": 1,
39
+ "prefill_chunk_size": 1024,
40
+ "sequence_length": 4096,
41
+ "speculation_length": 0,
42
+ "start_rank_id": 0,
43
+ "target": "trn1",
44
+ "torch_dtype": "bfloat16",
45
+ "tp_degree": 8
46
+ },
47
+ "num_attention_heads": 32,
48
+ "num_hidden_layers": 32,
49
+ "num_key_value_heads": 8,
50
+ "pretraining_tp": 1,
51
+ "rms_norm_eps": 1e-05,
52
+ "rope_scaling": {
53
+ "factor": 8.0,
54
+ "high_freq_factor": 4.0,
55
+ "low_freq_factor": 1.0,
56
+ "original_max_position_embeddings": 8192,
57
+ "rope_type": "llama3"
58
+ },
59
+ "rope_theta": 500000.0,
60
+ "tie_word_embeddings": false,
61
+ "use_cache": true,
62
+ "vocab_size": 128256
63
+ }
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/449164b2be0b8d13730f.json ADDED
@@ -0,0 +1,63 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_entry_class": "SingleModelCacheEntry",
3
+ "_model_id": "meta-llama/Llama-3.1-8B-Instruct",
4
+ "_task": "text-generation",
5
+ "architectures": [
6
+ "LlamaForCausalLM"
7
+ ],
8
+ "attention_bias": false,
9
+ "attention_dropout": 0.0,
10
+ "dtype": "bfloat16",
11
+ "head_dim": 128,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 4096,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 14336,
16
+ "max_position_embeddings": 131072,
17
+ "mlp_bias": false,
18
+ "model_type": "llama",
19
+ "neuron": {
20
+ "_serialized_key": "NxDNeuronConfig",
21
+ "batch_size": 32,
22
+ "capacity_factor": null,
23
+ "checkpoint_id": "meta-llama/Llama-3.1-8B-Instruct",
24
+ "checkpoint_revision": null,
25
+ "continuous_batching": true,
26
+ "ep_degree": 1,
27
+ "fused_qkv": true,
28
+ "glu_mlp": true,
29
+ "local_ranks_size": 8,
30
+ "max_batch_size": 32,
31
+ "max_context_length": 4096,
32
+ "max_topk": 256,
33
+ "n_active_tokens": 4096,
34
+ "neuronxcc_version": "2.21.33363.0+82129205",
35
+ "on_device_sampling": true,
36
+ "optimum_neuron_version": "0.4.6.dev1",
37
+ "output_logits": false,
38
+ "pp_degree": 1,
39
+ "prefill_chunk_size": 0,
40
+ "sequence_length": 4096,
41
+ "speculation_length": 0,
42
+ "start_rank_id": 0,
43
+ "target": "trn1",
44
+ "torch_dtype": "bfloat16",
45
+ "tp_degree": 8
46
+ },
47
+ "num_attention_heads": 32,
48
+ "num_hidden_layers": 32,
49
+ "num_key_value_heads": 8,
50
+ "pretraining_tp": 1,
51
+ "rms_norm_eps": 1e-05,
52
+ "rope_scaling": {
53
+ "factor": 8.0,
54
+ "high_freq_factor": 4.0,
55
+ "low_freq_factor": 1.0,
56
+ "original_max_position_embeddings": 8192,
57
+ "rope_type": "llama3"
58
+ },
59
+ "rope_theta": 500000.0,
60
+ "tie_word_embeddings": false,
61
+ "use_cache": true,
62
+ "vocab_size": 128256
63
+ }
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/5a9e3f4d03d538fb2f86.json ADDED
@@ -0,0 +1,63 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_entry_class": "SingleModelCacheEntry",
3
+ "_model_id": "meta-llama/Llama-3.1-8B-Instruct",
4
+ "_task": "text-generation",
5
+ "architectures": [
6
+ "LlamaForCausalLM"
7
+ ],
8
+ "attention_bias": false,
9
+ "attention_dropout": 0.0,
10
+ "dtype": "bfloat16",
11
+ "head_dim": 128,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 4096,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 14336,
16
+ "max_position_embeddings": 131072,
17
+ "mlp_bias": false,
18
+ "model_type": "llama",
19
+ "neuron": {
20
+ "_serialized_key": "NxDNeuronConfig",
21
+ "batch_size": 32,
22
+ "capacity_factor": null,
23
+ "checkpoint_id": "meta-llama/Llama-3.1-8B-Instruct",
24
+ "checkpoint_revision": null,
25
+ "continuous_batching": true,
26
+ "ep_degree": 1,
27
+ "fused_qkv": true,
28
+ "glu_mlp": true,
29
+ "local_ranks_size": 8,
30
+ "max_batch_size": 32,
31
+ "max_context_length": 4096,
32
+ "max_topk": 256,
33
+ "n_active_tokens": 4096,
34
+ "neuronxcc_version": "2.21.33363.0+82129205",
35
+ "on_device_sampling": false,
36
+ "optimum_neuron_version": "0.4.6.dev1",
37
+ "output_logits": false,
38
+ "pp_degree": 1,
39
+ "prefill_chunk_size": 0,
40
+ "sequence_length": 4096,
41
+ "speculation_length": 0,
42
+ "start_rank_id": 0,
43
+ "target": "trn1",
44
+ "torch_dtype": "bfloat16",
45
+ "tp_degree": 8
46
+ },
47
+ "num_attention_heads": 32,
48
+ "num_hidden_layers": 32,
49
+ "num_key_value_heads": 8,
50
+ "pretraining_tp": 1,
51
+ "rms_norm_eps": 1e-05,
52
+ "rope_scaling": {
53
+ "factor": 8.0,
54
+ "high_freq_factor": 4.0,
55
+ "low_freq_factor": 1.0,
56
+ "original_max_position_embeddings": 8192,
57
+ "rope_type": "llama3"
58
+ },
59
+ "rope_theta": 500000.0,
60
+ "tie_word_embeddings": false,
61
+ "use_cache": true,
62
+ "vocab_size": 128256
63
+ }
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev1/c54cbc73d9a74e547bf7ca1feb2b290b641ed261e32f7c08baba5633884f1298/7f7c6611f9515b012fbd.json ADDED
@@ -0,0 +1,63 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_entry_class": "SingleModelCacheEntry",
3
+ "_model_id": "meta-llama/Llama-3.1-8B-Instruct",
4
+ "_task": "text-generation",
5
+ "architectures": [
6
+ "LlamaForCausalLM"
7
+ ],
8
+ "attention_bias": false,
9
+ "attention_dropout": 0.0,
10
+ "dtype": "bfloat16",
11
+ "head_dim": 128,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 4096,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 14336,
16
+ "max_position_embeddings": 131072,
17
+ "mlp_bias": false,
18
+ "model_type": "llama",
19
+ "neuron": {
20
+ "_serialized_key": "NxDNeuronConfig",
21
+ "batch_size": 32,
22
+ "capacity_factor": null,
23
+ "checkpoint_id": "meta-llama/Llama-3.1-8B-Instruct",
24
+ "checkpoint_revision": "0e9e39f249a16976918f6564b8830bc894c89659",
25
+ "continuous_batching": true,
26
+ "ep_degree": 1,
27
+ "fused_qkv": true,
28
+ "glu_mlp": true,
29
+ "local_ranks_size": 8,
30
+ "max_batch_size": 32,
31
+ "max_context_length": 4096,
32
+ "max_topk": 256,
33
+ "n_active_tokens": 4096,
34
+ "neuronxcc_version": "2.21.33363.0+82129205",
35
+ "on_device_sampling": true,
36
+ "optimum_neuron_version": "0.4.6.dev1",
37
+ "output_logits": false,
38
+ "pp_degree": 1,
39
+ "prefill_chunk_size": 1024,
40
+ "sequence_length": 4096,
41
+ "speculation_length": 0,
42
+ "start_rank_id": 0,
43
+ "target": "trn1",
44
+ "torch_dtype": "bfloat16",
45
+ "tp_degree": 8
46
+ },
47
+ "num_attention_heads": 32,
48
+ "num_hidden_layers": 32,
49
+ "num_key_value_heads": 8,
50
+ "pretraining_tp": 1,
51
+ "rms_norm_eps": 1e-05,
52
+ "rope_scaling": {
53
+ "factor": 8.0,
54
+ "high_freq_factor": 4.0,
55
+ "low_freq_factor": 1.0,
56
+ "original_max_position_embeddings": 8192,
57
+ "rope_type": "llama3"
58
+ },
59
+ "rope_theta": 500000.0,
60
+ "tie_word_embeddings": false,
61
+ "use_cache": true,
62
+ "vocab_size": 128256
63
+ }
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/3b53548b380293de28b8.json ADDED
@@ -0,0 +1,64 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_entry_class": "SingleModelCacheEntry",
3
+ "_model_id": "unsloth/Llama-3.2-1B-Instruct",
4
+ "_task": "text-generation",
5
+ "architectures": [
6
+ "LlamaForCausalLM"
7
+ ],
8
+ "attention_bias": false,
9
+ "attention_dropout": 0.0,
10
+ "dtype": "bfloat16",
11
+ "head_dim": 64,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 2048,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 8192,
16
+ "max_position_embeddings": 131072,
17
+ "mlp_bias": false,
18
+ "model_type": "llama",
19
+ "neuron": {
20
+ "_serialized_key": "NxDNeuronConfig",
21
+ "batch_size": 2,
22
+ "capacity_factor": null,
23
+ "checkpoint_id": "unsloth/Llama-3.2-1B-Instruct",
24
+ "checkpoint_revision": null,
25
+ "continuous_batching": false,
26
+ "ep_degree": 1,
27
+ "fused_qkv": true,
28
+ "glu_mlp": true,
29
+ "local_ranks_size": 2,
30
+ "max_batch_size": 2,
31
+ "max_context_length": 4096,
32
+ "max_topk": 256,
33
+ "n_active_tokens": 4096,
34
+ "neuronxcc_version": "2.21.33363.0+82129205",
35
+ "on_device_sampling": false,
36
+ "optimum_neuron_version": "0.4.6.dev2",
37
+ "output_logits": false,
38
+ "pp_degree": 1,
39
+ "prefill_chunk_size": 0,
40
+ "sequence_length": 4096,
41
+ "speculation_length": 0,
42
+ "start_rank_id": 0,
43
+ "target": "trn1",
44
+ "torch_dtype": "bfloat16",
45
+ "tp_degree": 2
46
+ },
47
+ "num_attention_heads": 32,
48
+ "num_hidden_layers": 16,
49
+ "num_key_value_heads": 8,
50
+ "pretraining_tp": 1,
51
+ "rms_norm_eps": 1e-05,
52
+ "rope_scaling": {
53
+ "factor": 32.0,
54
+ "high_freq_factor": 4.0,
55
+ "low_freq_factor": 1.0,
56
+ "original_max_position_embeddings": 8192,
57
+ "rope_type": "llama3"
58
+ },
59
+ "rope_theta": 500000.0,
60
+ "tie_word_embeddings": true,
61
+ "unsloth_fixed": true,
62
+ "use_cache": true,
63
+ "vocab_size": 128256
64
+ }
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/700b41cac5912175b08c.json ADDED
@@ -0,0 +1,64 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_entry_class": "SingleModelCacheEntry",
3
+ "_model_id": "unsloth/Llama-3.2-1B-Instruct",
4
+ "_task": "text-generation",
5
+ "architectures": [
6
+ "LlamaForCausalLM"
7
+ ],
8
+ "attention_bias": false,
9
+ "attention_dropout": 0.0,
10
+ "dtype": "bfloat16",
11
+ "head_dim": 64,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 2048,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 8192,
16
+ "max_position_embeddings": 131072,
17
+ "mlp_bias": false,
18
+ "model_type": "llama",
19
+ "neuron": {
20
+ "_serialized_key": "NxDNeuronConfig",
21
+ "batch_size": 1,
22
+ "capacity_factor": null,
23
+ "checkpoint_id": "unsloth/Llama-3.2-1B-Instruct",
24
+ "checkpoint_revision": null,
25
+ "continuous_batching": false,
26
+ "ep_degree": 1,
27
+ "fused_qkv": true,
28
+ "glu_mlp": true,
29
+ "local_ranks_size": 2,
30
+ "max_batch_size": 1,
31
+ "max_context_length": 512,
32
+ "max_topk": 256,
33
+ "n_active_tokens": 512,
34
+ "neuronxcc_version": "2.21.33363.0+82129205",
35
+ "on_device_sampling": false,
36
+ "optimum_neuron_version": "0.4.6.dev2",
37
+ "output_logits": false,
38
+ "pp_degree": 1,
39
+ "prefill_chunk_size": 0,
40
+ "sequence_length": 512,
41
+ "speculation_length": 0,
42
+ "start_rank_id": 0,
43
+ "target": "trn1",
44
+ "torch_dtype": "bfloat16",
45
+ "tp_degree": 2
46
+ },
47
+ "num_attention_heads": 32,
48
+ "num_hidden_layers": 16,
49
+ "num_key_value_heads": 8,
50
+ "pretraining_tp": 1,
51
+ "rms_norm_eps": 1e-05,
52
+ "rope_scaling": {
53
+ "factor": 32.0,
54
+ "high_freq_factor": 4.0,
55
+ "low_freq_factor": 1.0,
56
+ "original_max_position_embeddings": 8192,
57
+ "rope_type": "llama3"
58
+ },
59
+ "rope_theta": 500000.0,
60
+ "tie_word_embeddings": true,
61
+ "unsloth_fixed": true,
62
+ "use_cache": true,
63
+ "vocab_size": 128256
64
+ }
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/cf6b9a360dcf294104671106bae2adbd9fd291823bb60a351883163684073231/792c553a46127d9f996b.json ADDED
@@ -0,0 +1,64 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_entry_class": "SingleModelCacheEntry",
3
+ "_model_id": "unsloth/Llama-3.2-1B-Instruct",
4
+ "_task": "text-generation",
5
+ "architectures": [
6
+ "LlamaForCausalLM"
7
+ ],
8
+ "attention_bias": false,
9
+ "attention_dropout": 0.0,
10
+ "dtype": "bfloat16",
11
+ "head_dim": 64,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 2048,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 8192,
16
+ "max_position_embeddings": 131072,
17
+ "mlp_bias": false,
18
+ "model_type": "llama",
19
+ "neuron": {
20
+ "_serialized_key": "NxDNeuronConfig",
21
+ "batch_size": 1,
22
+ "capacity_factor": null,
23
+ "checkpoint_id": "unsloth/Llama-3.2-1B-Instruct",
24
+ "checkpoint_revision": null,
25
+ "continuous_batching": false,
26
+ "ep_degree": 1,
27
+ "fused_qkv": true,
28
+ "glu_mlp": true,
29
+ "local_ranks_size": 2,
30
+ "max_batch_size": 1,
31
+ "max_context_length": 512,
32
+ "max_topk": 256,
33
+ "n_active_tokens": 512,
34
+ "neuronxcc_version": "2.21.33363.0+82129205",
35
+ "on_device_sampling": false,
36
+ "optimum_neuron_version": "0.4.6.dev2",
37
+ "output_logits": false,
38
+ "pp_degree": 1,
39
+ "prefill_chunk_size": 64,
40
+ "sequence_length": 512,
41
+ "speculation_length": 0,
42
+ "start_rank_id": 0,
43
+ "target": "trn1",
44
+ "torch_dtype": "bfloat16",
45
+ "tp_degree": 2
46
+ },
47
+ "num_attention_heads": 32,
48
+ "num_hidden_layers": 16,
49
+ "num_key_value_heads": 8,
50
+ "pretraining_tp": 1,
51
+ "rms_norm_eps": 1e-05,
52
+ "rope_scaling": {
53
+ "factor": 32.0,
54
+ "high_freq_factor": 4.0,
55
+ "low_freq_factor": 1.0,
56
+ "original_max_position_embeddings": 8192,
57
+ "rope_type": "llama3"
58
+ },
59
+ "rope_theta": 500000.0,
60
+ "tie_word_embeddings": true,
61
+ "unsloth_fixed": true,
62
+ "use_cache": true,
63
+ "vocab_size": 128256
64
+ }
neuronxcc-2.21.33363.0+82129205/0_REGISTRY/0.4.6.dev2/llama/unsloth/Llama-3.2-1B-Instruct/700b41cac5912175b08c.json ADDED
@@ -0,0 +1,64 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_entry_class": "SingleModelCacheEntry",
3
+ "_model_id": "unsloth/Llama-3.2-1B-Instruct",
4
+ "_task": "text-generation",
5
+ "architectures": [
6
+ "LlamaForCausalLM"
7
+ ],
8
+ "attention_bias": false,
9
+ "attention_dropout": 0.0,
10
+ "dtype": "bfloat16",
11
+ "head_dim": 64,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 2048,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 8192,
16
+ "max_position_embeddings": 131072,
17
+ "mlp_bias": false,
18
+ "model_type": "llama",
19
+ "neuron": {
20
+ "_serialized_key": "NxDNeuronConfig",
21
+ "batch_size": 1,
22
+ "capacity_factor": null,
23
+ "checkpoint_id": "unsloth/Llama-3.2-1B-Instruct",
24
+ "checkpoint_revision": null,
25
+ "continuous_batching": false,
26
+ "ep_degree": 1,
27
+ "fused_qkv": true,
28
+ "glu_mlp": true,
29
+ "local_ranks_size": 2,
30
+ "max_batch_size": 1,
31
+ "max_context_length": 512,
32
+ "max_topk": 256,
33
+ "n_active_tokens": 512,
34
+ "neuronxcc_version": "2.21.33363.0+82129205",
35
+ "on_device_sampling": false,
36
+ "optimum_neuron_version": "0.4.6.dev2",
37
+ "output_logits": false,
38
+ "pp_degree": 1,
39
+ "prefill_chunk_size": 0,
40
+ "sequence_length": 512,
41
+ "speculation_length": 0,
42
+ "start_rank_id": 0,
43
+ "target": "trn1",
44
+ "torch_dtype": "bfloat16",
45
+ "tp_degree": 2
46
+ },
47
+ "num_attention_heads": 32,
48
+ "num_hidden_layers": 16,
49
+ "num_key_value_heads": 8,
50
+ "pretraining_tp": 1,
51
+ "rms_norm_eps": 1e-05,
52
+ "rope_scaling": {
53
+ "factor": 32.0,
54
+ "high_freq_factor": 4.0,
55
+ "low_freq_factor": 1.0,
56
+ "original_max_position_embeddings": 8192,
57
+ "rope_type": "llama3"
58
+ },
59
+ "rope_theta": 500000.0,
60
+ "tie_word_embeddings": true,
61
+ "unsloth_fixed": true,
62
+ "use_cache": true,
63
+ "vocab_size": 128256
64
+ }
neuronxcc-2.21.33363.0+82129205/MODULE_15104978417860996248+e30acd3a/model.neff CHANGED
Binary files a/neuronxcc-2.21.33363.0+82129205/MODULE_15104978417860996248+e30acd3a/model.neff and b/neuronxcc-2.21.33363.0+82129205/MODULE_15104978417860996248+e30acd3a/model.neff differ
 
neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/compile_flags.json ADDED
@@ -0,0 +1 @@
 
 
1
+ ["--target=trn1", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "-O2", "--lnc=1", "--logfile=/tmp/nxd_model/token_generation/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"]
neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/model.done ADDED
File without changes
neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/model.hlo_module.pb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fc384fc0324b11db5da96604331cd5055d6c75432751e5e40d6a8365b8fa648a
3
+ size 799960
neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/model.neff ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a40ec25bcacdb7caad8ec72238d804a702ac2b36d50f2eae5452615372333e55
3
+ size 6872064
neuronxcc-2.21.33363.0+82129205/MODULE_1bfeeed3b11ee15a7e41+a02c3a36/wrapped_neff.hlo ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ccb46abb3a38777c337bac220a060af8fe90ec103f8eb1d8e99a288592f166d6
3
+ size 7019528
neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/compile_flags.json ADDED
@@ -0,0 +1 @@
 
 
1
+ ["--target=trn1", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "-O2", "--lnc=1", "--logfile=/tmp/nxd_model/chunked_prefill/_tp0_bk0/log-neuron-cc.txt"]
neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/model.done ADDED
File without changes
neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/model.hlo_module.pb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f26bf8ee45db3f0f1962d5413d680ac63ccb89bc66ab6282e0e02415fa8f5e2c
3
+ size 498827
neuronxcc-2.21.33363.0+82129205/MODULE_483d6170aeb3a34fef2e+6170d8e1/model.neff ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:674273a150bfda8e7977bae1985e99a62ef564db8572385d157240ffe7600813
3
+ size 2120704
neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/compile_flags.json ADDED
@@ -0,0 +1 @@
 
 
1
+ ["--target=trn1", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "-O2", "--lnc=1", "--logfile=/tmp/nxd_model/token_generation/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"]
neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/model.done ADDED
File without changes
neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/model.hlo_module.pb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:206450a370d213a0ad9a7eb0ddfd4050292bf64ecf17344b02204c661dcfabf3
3
+ size 397298
neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/model.neff ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1aae8abec5aee4c1fcae30792817224d16127cd62c27bec469cd8f9802bd1dbd
3
+ size 1741824
neuronxcc-2.21.33363.0+82129205/MODULE_5cc25de5cd6d6bbd374a+a02c3a36/wrapped_neff.hlo ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4852707561e1f28060ba885c808186aeecdf0c1389b0417de838d09cce390175
3
+ size 1815793
neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/compile_flags.json ADDED
@@ -0,0 +1 @@
 
 
1
+ ["--target=trn1", "--enable-saturate-infinity", "--auto-cast=none", "--model-type=transformer", "-O1", "--logfile=/tmp/nxdi_test_48ffe586-34f9-4592-bf76-89fbf67ad5e6/compiler_workdir/SoftmaxNoMask/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"]
neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/model.done ADDED
File without changes
neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/model.hlo_module.pb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8e542c76cf27c3f807feac47dd3f58fd4c308e8d9e932f0732774bee19d640d8
3
+ size 3881
neuronxcc-2.21.33363.0+82129205/MODULE_89cea99948e25972a914+53af71ad/model.neff ADDED
Binary file (31.7 kB). View file
 
neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/compile_flags.json ADDED
@@ -0,0 +1 @@
 
 
1
+ ["--target=trn1", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "-O2", "--lnc=1", "--logfile=/tmp/nxd_model/context_encoding/_tp0_bk0/log-neuron-cc.txt"]
neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/model.done ADDED
File without changes
neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/model.hlo_module.pb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a945df4dd154fab077f1da24bde6fb2e46aa20de51c6cc39d5545ffb7c64d733
3
+ size 337649
neuronxcc-2.21.33363.0+82129205/MODULE_8a43fc03fdbaa14ac17d+24129607/model.neff ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0bcae3ba218cc15ae1709d588223e5a1c5e2a512a6d5156146c0cbafdec18162
3
+ size 2612224
neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/compile_flags.json ADDED
@@ -0,0 +1 @@
 
 
1
+ ["--target=trn1", "--enable-saturate-infinity", "--auto-cast=none", "--model-type=transformer", "-O1", "--logfile=/tmp/nxdi_test_75ae5c33-5e1e-46c6-8477-19333053591f/compiler_workdir/SoftmaxNoMask/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"]
neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/model.done ADDED
File without changes
neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/model.hlo_module.pb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5efb0ea01dc0241314f91e19cbd9be551d1a66e7dbadd7cd4d6473d6b7d4b99a
3
+ size 3881
neuronxcc-2.21.33363.0+82129205/MODULE_9d60ee120c3f9e8fc639+d93d7f4a/model.neff ADDED
Binary file (31.7 kB). View file
 
neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/compile_flags.json ADDED
@@ -0,0 +1 @@
 
 
1
+ ["--target=trn1", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "-O2", "--lnc=1", "--logfile=/tmp/nxd_model/chunked_prefill/_tp0_bk0/log-neuron-cc.txt"]
neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/model.done ADDED
File without changes
neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/model.hlo_module.pb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3296a30cf85cdc9498292d7213649898ef13ed89a3910ca123d7526eb9824e97
3
+ size 1034416
neuronxcc-2.21.33363.0+82129205/MODULE_b3e6b43fde45aac4a200+6170d8e1/model.neff ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6fa161991c7b70c2283cfd5351df193872e23bc89b0d68deca4e6fc5a6b74bb9
3
+ size 18996224
neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/compile_flags.json ADDED
@@ -0,0 +1 @@
 
 
1
+ ["--target=trn1", "--enable-saturate-infinity", "--auto-cast=none", "--model-type=transformer", "-O1", "--logfile=/tmp/nxdi_test_bde15d7b-1637-4861-8418-410426f6fc99/compiler_workdir/SoftmaxWithMask/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"]
neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/model.done ADDED
File without changes
neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/model.hlo_module.pb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:70f96487dad4bb02b98bf2c955fe59650a5fdbcf1d763fdf56ec412b62b5774c
3
+ size 5596
neuronxcc-2.21.33363.0+82129205/MODULE_b6a1e5451876174f07da+a604d091/model.neff ADDED
Binary file (31.7 kB). View file
 
neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/compile_flags.json ADDED
@@ -0,0 +1 @@
 
 
1
+ ["--target=trn1", "--auto-cast=none", "--model-type=transformer", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2 --vectorize-strided-dma ", "-O2", "--lnc=1", "--logfile=/tmp/nxd_model/context_encoding/_tp0_bk0/log-neuron-cc.txt"]
neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/model.done ADDED
File without changes
neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/model.hlo_module.pb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8eb71fdd81d66f5a362985d7ac7802aa3739bc1d902de7ecff7b933bf65bfcec
3
+ size 861207
neuronxcc-2.21.33363.0+82129205/MODULE_b6da35fd5e809d42ec50+24129607/model.neff ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fdff315f0a385fe46b91a0fc08f91f35ddd537601a850735ca27314c468efc9d
3
+ size 60457984