dacorvo HF Staff commited on
Commit
927c7bb
·
verified ·
1 Parent(s): 6fd1445

Synchronizing local compiler cache.

Browse files
Files changed (22) hide show
  1. .gitattributes +6 -0
  2. neuronxcc-2.19.8089.0+8ab9f450/0_REGISTRY/0.3.1.dev6/llama/unsloth/Llama-3.1-8B-Instruct/6bb470a0ab2cc05af9d7.json +64 -0
  3. neuronxcc-2.19.8089.0+8ab9f450/0_REGISTRY/0.3.1.dev6/qwen3_moe/Qwen/Qwen3-30B-A3B-Instruct-2507/74cd3e6ddd109ea79fae.json +66 -0
  4. neuronxcc-2.19.8089.0+8ab9f450/0_REGISTRY/0.3.1.dev6/qwen3_moe/Qwen/Qwen3-30B-A3B-Instruct-2507/b3ddf1f20b6d95203808.json +66 -0
  5. neuronxcc-2.19.8089.0+8ab9f450/MODULE_0e124b1b98cdfaaea978+253d6470/compile_flags.json +1 -0
  6. neuronxcc-2.19.8089.0+8ab9f450/MODULE_0e124b1b98cdfaaea978+253d6470/model.done +0 -0
  7. neuronxcc-2.19.8089.0+8ab9f450/MODULE_0e124b1b98cdfaaea978+253d6470/model.hlo_module.pb +3 -0
  8. neuronxcc-2.19.8089.0+8ab9f450/MODULE_0e124b1b98cdfaaea978+253d6470/model.neff +3 -0
  9. neuronxcc-2.19.8089.0+8ab9f450/MODULE_aabe99e7dc32f98f103f+cd3419b6/compile_flags.json +1 -0
  10. neuronxcc-2.19.8089.0+8ab9f450/MODULE_aabe99e7dc32f98f103f+cd3419b6/model.done +0 -0
  11. neuronxcc-2.19.8089.0+8ab9f450/MODULE_aabe99e7dc32f98f103f+cd3419b6/model.hlo_module.pb +3 -0
  12. neuronxcc-2.19.8089.0+8ab9f450/MODULE_aabe99e7dc32f98f103f+cd3419b6/model.neff +3 -0
  13. neuronxcc-2.19.8089.0+8ab9f450/MODULE_aabe99e7dc32f98f103f+cd3419b6/wrapped_neff.hlo +3 -0
  14. neuronxcc-2.19.8089.0+8ab9f450/MODULE_ceaea2495f617f265d2c+cd3419b6/compile_flags.json +1 -0
  15. neuronxcc-2.19.8089.0+8ab9f450/MODULE_ceaea2495f617f265d2c+cd3419b6/model.done +0 -0
  16. neuronxcc-2.19.8089.0+8ab9f450/MODULE_ceaea2495f617f265d2c+cd3419b6/model.hlo_module.pb +3 -0
  17. neuronxcc-2.19.8089.0+8ab9f450/MODULE_ceaea2495f617f265d2c+cd3419b6/model.neff +3 -0
  18. neuronxcc-2.19.8089.0+8ab9f450/MODULE_ceaea2495f617f265d2c+cd3419b6/wrapped_neff.hlo +3 -0
  19. neuronxcc-2.19.8089.0+8ab9f450/MODULE_ef05c748e5821ca55683+253d6470/compile_flags.json +1 -0
  20. neuronxcc-2.19.8089.0+8ab9f450/MODULE_ef05c748e5821ca55683+253d6470/model.done +0 -0
  21. neuronxcc-2.19.8089.0+8ab9f450/MODULE_ef05c748e5821ca55683+253d6470/model.hlo_module.pb +3 -0
  22. neuronxcc-2.19.8089.0+8ab9f450/MODULE_ef05c748e5821ca55683+253d6470/model.neff +3 -0
.gitattributes CHANGED
@@ -10988,3 +10988,9 @@ neuronxcc-2.19.8089.0+8ab9f450/MODULE_630352802875996106+5860ac52/model.neff fil
10988
  neuronxcc-2.19.8089.0+8ab9f450/MODULE_670328823403641946+5860ac52/model.neff filter=lfs diff=lfs merge=lfs -text
10989
  neuronxcc-2.19.8089.0+8ab9f450/MODULE_7714349839539442872+5860ac52/model.neff filter=lfs diff=lfs merge=lfs -text
10990
  neuronxcc-2.19.8089.0+8ab9f450/MODULE_8000038047450194595+5860ac52/model.neff filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
10988
  neuronxcc-2.19.8089.0+8ab9f450/MODULE_670328823403641946+5860ac52/model.neff filter=lfs diff=lfs merge=lfs -text
10989
  neuronxcc-2.19.8089.0+8ab9f450/MODULE_7714349839539442872+5860ac52/model.neff filter=lfs diff=lfs merge=lfs -text
10990
  neuronxcc-2.19.8089.0+8ab9f450/MODULE_8000038047450194595+5860ac52/model.neff filter=lfs diff=lfs merge=lfs -text
10991
+ neuronxcc-2.19.8089.0+8ab9f450/MODULE_0e124b1b98cdfaaea978+253d6470/model.neff filter=lfs diff=lfs merge=lfs -text
10992
+ neuronxcc-2.19.8089.0+8ab9f450/MODULE_aabe99e7dc32f98f103f+cd3419b6/model.neff filter=lfs diff=lfs merge=lfs -text
10993
+ neuronxcc-2.19.8089.0+8ab9f450/MODULE_aabe99e7dc32f98f103f+cd3419b6/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
10994
+ neuronxcc-2.19.8089.0+8ab9f450/MODULE_ceaea2495f617f265d2c+cd3419b6/model.neff filter=lfs diff=lfs merge=lfs -text
10995
+ neuronxcc-2.19.8089.0+8ab9f450/MODULE_ceaea2495f617f265d2c+cd3419b6/wrapped_neff.hlo filter=lfs diff=lfs merge=lfs -text
10996
+ neuronxcc-2.19.8089.0+8ab9f450/MODULE_ef05c748e5821ca55683+253d6470/model.neff filter=lfs diff=lfs merge=lfs -text
neuronxcc-2.19.8089.0+8ab9f450/0_REGISTRY/0.3.1.dev6/llama/unsloth/Llama-3.1-8B-Instruct/6bb470a0ab2cc05af9d7.json ADDED
@@ -0,0 +1,64 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_entry_class": "SingleModelCacheEntry",
3
+ "_model_id": "unsloth/Llama-3.1-8B-Instruct",
4
+ "_task": "text-generation",
5
+ "architectures": [
6
+ "LlamaForCausalLM"
7
+ ],
8
+ "attention_bias": false,
9
+ "attention_dropout": 0.0,
10
+ "head_dim": 128,
11
+ "hidden_act": "silu",
12
+ "hidden_size": 4096,
13
+ "initializer_range": 0.02,
14
+ "intermediate_size": 14336,
15
+ "max_position_embeddings": 131072,
16
+ "mlp_bias": false,
17
+ "model_type": "llama",
18
+ "neuron": {
19
+ "_serialized_key": "NxDNeuronConfig",
20
+ "batch_size": 32,
21
+ "capacity_factor": null,
22
+ "checkpoint_id": "unsloth/Llama-3.1-8B-Instruct",
23
+ "checkpoint_revision": "4699cc75b550f9c6f3173fb80f4703b62d946aa5",
24
+ "continuous_batching": true,
25
+ "enable_bucketing": false,
26
+ "ep_degree": 1,
27
+ "fused_qkv": true,
28
+ "glu_mlp": true,
29
+ "local_ranks_size": 8,
30
+ "logical_nc_config": 1,
31
+ "max_batch_size": 32,
32
+ "max_context_length": 4096,
33
+ "max_topk": 256,
34
+ "n_active_tokens": 4096,
35
+ "neuronxcc_version": "2.19.8089.0+8ab9f450",
36
+ "on_device_sampling": true,
37
+ "optimum_neuron_version": "0.3.1.dev6",
38
+ "output_logits": false,
39
+ "pp_degree": 1,
40
+ "sequence_length": 4096,
41
+ "speculation_length": 0,
42
+ "start_rank_id": 0,
43
+ "target": null,
44
+ "torch_dtype": "bfloat16",
45
+ "tp_degree": 8
46
+ },
47
+ "num_attention_heads": 32,
48
+ "num_hidden_layers": 32,
49
+ "num_key_value_heads": 8,
50
+ "pretraining_tp": 1,
51
+ "rms_norm_eps": 1e-05,
52
+ "rope_scaling": {
53
+ "factor": 8.0,
54
+ "high_freq_factor": 4.0,
55
+ "low_freq_factor": 1.0,
56
+ "original_max_position_embeddings": 8192,
57
+ "rope_type": "llama3"
58
+ },
59
+ "rope_theta": 500000.0,
60
+ "tie_word_embeddings": false,
61
+ "unsloth_fixed": true,
62
+ "use_cache": true,
63
+ "vocab_size": 128256
64
+ }
neuronxcc-2.19.8089.0+8ab9f450/0_REGISTRY/0.3.1.dev6/qwen3_moe/Qwen/Qwen3-30B-A3B-Instruct-2507/74cd3e6ddd109ea79fae.json ADDED
@@ -0,0 +1,66 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_entry_class": "SingleModelCacheEntry",
3
+ "_model_id": "Qwen/Qwen3-30B-A3B-Instruct-2507",
4
+ "_task": "text-generation",
5
+ "architectures": [
6
+ "Qwen3MoeForCausalLM"
7
+ ],
8
+ "attention_bias": false,
9
+ "attention_dropout": 0.0,
10
+ "decoder_sparse_step": 1,
11
+ "head_dim": 128,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 2048,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 6144,
16
+ "max_position_embeddings": 262144,
17
+ "max_window_layers": 48,
18
+ "mlp_only_layers": [],
19
+ "model_type": "qwen3_moe",
20
+ "moe_intermediate_size": 768,
21
+ "neuron": {
22
+ "_serialized_key": "NxDNeuronConfig",
23
+ "batch_size": 8,
24
+ "capacity_factor": null,
25
+ "checkpoint_id": "Qwen/Qwen3-30B-A3B-Instruct-2507",
26
+ "checkpoint_revision": "0d7cf23991f47feeb3a57ecb4c9cee8ea4a17bfe",
27
+ "continuous_batching": true,
28
+ "enable_bucketing": false,
29
+ "ep_degree": 1,
30
+ "fused_qkv": false,
31
+ "glu_mlp": true,
32
+ "local_ranks_size": 32,
33
+ "logical_nc_config": 1,
34
+ "max_batch_size": 8,
35
+ "max_context_length": 4096,
36
+ "max_topk": 256,
37
+ "n_active_tokens": 4096,
38
+ "neuronxcc_version": "2.19.8089.0+8ab9f450",
39
+ "on_device_sampling": true,
40
+ "optimum_neuron_version": "0.3.1.dev6",
41
+ "output_logits": false,
42
+ "pp_degree": 1,
43
+ "sequence_length": 4096,
44
+ "speculation_length": 0,
45
+ "start_rank_id": 0,
46
+ "target": null,
47
+ "torch_dtype": "bfloat16",
48
+ "tp_degree": 32
49
+ },
50
+ "norm_topk_prob": true,
51
+ "num_attention_heads": 32,
52
+ "num_experts": 128,
53
+ "num_experts_per_tok": 8,
54
+ "num_hidden_layers": 48,
55
+ "num_key_value_heads": 4,
56
+ "output_router_logits": false,
57
+ "rms_norm_eps": 1e-06,
58
+ "rope_scaling": null,
59
+ "rope_theta": 10000000,
60
+ "router_aux_loss_coef": 0.001,
61
+ "sliding_window": null,
62
+ "tie_word_embeddings": false,
63
+ "use_cache": true,
64
+ "use_sliding_window": false,
65
+ "vocab_size": 151936
66
+ }
neuronxcc-2.19.8089.0+8ab9f450/0_REGISTRY/0.3.1.dev6/qwen3_moe/Qwen/Qwen3-30B-A3B-Instruct-2507/b3ddf1f20b6d95203808.json ADDED
@@ -0,0 +1,66 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_entry_class": "SingleModelCacheEntry",
3
+ "_model_id": "Qwen/Qwen3-30B-A3B-Instruct-2507",
4
+ "_task": "text-generation",
5
+ "architectures": [
6
+ "Qwen3MoeForCausalLM"
7
+ ],
8
+ "attention_bias": false,
9
+ "attention_dropout": 0.0,
10
+ "decoder_sparse_step": 1,
11
+ "head_dim": 128,
12
+ "hidden_act": "silu",
13
+ "hidden_size": 2048,
14
+ "initializer_range": 0.02,
15
+ "intermediate_size": 6144,
16
+ "max_position_embeddings": 262144,
17
+ "max_window_layers": 48,
18
+ "mlp_only_layers": [],
19
+ "model_type": "qwen3_moe",
20
+ "moe_intermediate_size": 768,
21
+ "neuron": {
22
+ "_serialized_key": "NxDNeuronConfig",
23
+ "batch_size": 16,
24
+ "capacity_factor": null,
25
+ "checkpoint_id": "Qwen/Qwen3-30B-A3B-Instruct-2507",
26
+ "checkpoint_revision": "0d7cf23991f47feeb3a57ecb4c9cee8ea4a17bfe",
27
+ "continuous_batching": true,
28
+ "enable_bucketing": false,
29
+ "ep_degree": 1,
30
+ "fused_qkv": false,
31
+ "glu_mlp": true,
32
+ "local_ranks_size": 32,
33
+ "logical_nc_config": 1,
34
+ "max_batch_size": 16,
35
+ "max_context_length": 4096,
36
+ "max_topk": 256,
37
+ "n_active_tokens": 4096,
38
+ "neuronxcc_version": "2.19.8089.0+8ab9f450",
39
+ "on_device_sampling": true,
40
+ "optimum_neuron_version": "0.3.1.dev6",
41
+ "output_logits": false,
42
+ "pp_degree": 1,
43
+ "sequence_length": 4096,
44
+ "speculation_length": 0,
45
+ "start_rank_id": 0,
46
+ "target": null,
47
+ "torch_dtype": "bfloat16",
48
+ "tp_degree": 32
49
+ },
50
+ "norm_topk_prob": true,
51
+ "num_attention_heads": 32,
52
+ "num_experts": 128,
53
+ "num_experts_per_tok": 8,
54
+ "num_hidden_layers": 48,
55
+ "num_key_value_heads": 4,
56
+ "output_router_logits": false,
57
+ "rms_norm_eps": 1e-06,
58
+ "rope_scaling": null,
59
+ "rope_theta": 10000000,
60
+ "router_aux_loss_coef": 0.001,
61
+ "sliding_window": null,
62
+ "tie_word_embeddings": false,
63
+ "use_cache": true,
64
+ "use_sliding_window": false,
65
+ "vocab_size": 151936
66
+ }
neuronxcc-2.19.8089.0+8ab9f450/MODULE_0e124b1b98cdfaaea978+253d6470/compile_flags.json ADDED
@@ -0,0 +1 @@
 
 
1
+ ["--target=trn1", "--enable-saturate-infinity", "--enable-mixed-precision-accumulation", "--model-type", "transformer", "-O1", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2", "--auto-cast=none", "--internal-enable-dge-levels", "vector_dynamic_offsets", "--internal-hlo2tensorizer-options=--verify-hlo=true", "--logfile=/tmp/nxd_model/context_encoding_model/_tp0_bk0/log-neuron-cc.txt"]
neuronxcc-2.19.8089.0+8ab9f450/MODULE_0e124b1b98cdfaaea978+253d6470/model.done ADDED
File without changes
neuronxcc-2.19.8089.0+8ab9f450/MODULE_0e124b1b98cdfaaea978+253d6470/model.hlo_module.pb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8b08ca8fd26e8777354c5b1071686a7c1a982bc6e75e22b015204f20f059784f
3
+ size 9577957
neuronxcc-2.19.8089.0+8ab9f450/MODULE_0e124b1b98cdfaaea978+253d6470/model.neff ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eed5c37ecf28858415088c05f4683d4fba9aa0dcb67f7c71bab5b8d1d93596d0
3
+ size 17132544
neuronxcc-2.19.8089.0+8ab9f450/MODULE_aabe99e7dc32f98f103f+cd3419b6/compile_flags.json ADDED
@@ -0,0 +1 @@
 
 
1
+ ["--target=trn1", "--enable-saturate-infinity", "--enable-mixed-precision-accumulation", "--model-type", "transformer", "-O1", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2", "--auto-cast=none", "--internal-enable-dge-levels", "vector_dynamic_offsets", "--internal-hlo2tensorizer-options=--verify-hlo=true", "--logfile=/tmp/nxd_model/token_generation_model/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"]
neuronxcc-2.19.8089.0+8ab9f450/MODULE_aabe99e7dc32f98f103f+cd3419b6/model.done ADDED
File without changes
neuronxcc-2.19.8089.0+8ab9f450/MODULE_aabe99e7dc32f98f103f+cd3419b6/model.hlo_module.pb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c20b1ba450a3e0d7e2f46a4b36c5463c48b0a45b5d03084b2ec72af3b87edd0a
3
+ size 8763637
neuronxcc-2.19.8089.0+8ab9f450/MODULE_aabe99e7dc32f98f103f+cd3419b6/model.neff ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:011bdf691ace7030f1fb1026a583b5efbbeecdd9b0c956606bc1eeb2308b4c72
3
+ size 9534464
neuronxcc-2.19.8089.0+8ab9f450/MODULE_aabe99e7dc32f98f103f+cd3419b6/wrapped_neff.hlo ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0eda2f3ac2af2d06983b3855d6a6263911be0667649fd8d3e1c2814ff9d562d9
3
+ size 9804001
neuronxcc-2.19.8089.0+8ab9f450/MODULE_ceaea2495f617f265d2c+cd3419b6/compile_flags.json ADDED
@@ -0,0 +1 @@
 
 
1
+ ["--target=trn1", "--enable-saturate-infinity", "--enable-mixed-precision-accumulation", "--model-type", "transformer", "-O1", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2", "--auto-cast=none", "--internal-enable-dge-levels", "vector_dynamic_offsets", "--internal-hlo2tensorizer-options=--verify-hlo=true", "--logfile=/tmp/nxd_model/token_generation_model/_tp0_bk0/log-neuron-cc.txt", "--enable-internal-neff-wrapper"]
neuronxcc-2.19.8089.0+8ab9f450/MODULE_ceaea2495f617f265d2c+cd3419b6/model.done ADDED
File without changes
neuronxcc-2.19.8089.0+8ab9f450/MODULE_ceaea2495f617f265d2c+cd3419b6/model.hlo_module.pb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:077c0ac570b2a0c16265de87d80bdbed05adb4dd9dc43dc491c3cf44134146d0
3
+ size 3362881
neuronxcc-2.19.8089.0+8ab9f450/MODULE_ceaea2495f617f265d2c+cd3419b6/model.neff ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b3b0de5974a4f94bbe5a4be90ea97e78e0201bfe58fb07a56f8693704f84a5d8
3
+ size 6022144
neuronxcc-2.19.8089.0+8ab9f450/MODULE_ceaea2495f617f265d2c+cd3419b6/wrapped_neff.hlo ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1bec099529a73e12c52f26be6571e59bcea1bfb3ac59bc80283371a37dffce79
3
+ size 6270798
neuronxcc-2.19.8089.0+8ab9f450/MODULE_ef05c748e5821ca55683+253d6470/compile_flags.json ADDED
@@ -0,0 +1 @@
 
 
1
+ ["--target=trn1", "--enable-saturate-infinity", "--enable-mixed-precision-accumulation", "--model-type", "transformer", "-O1", "--tensorizer-options=--enable-ccop-compute-overlap --cc-pipeline-tiling-factor=2", "--auto-cast=none", "--internal-enable-dge-levels", "vector_dynamic_offsets", "--internal-hlo2tensorizer-options=--verify-hlo=true", "--logfile=/tmp/nxd_model/context_encoding_model/_tp0_bk0/log-neuron-cc.txt"]
neuronxcc-2.19.8089.0+8ab9f450/MODULE_ef05c748e5821ca55683+253d6470/model.done ADDED
File without changes
neuronxcc-2.19.8089.0+8ab9f450/MODULE_ef05c748e5821ca55683+253d6470/model.hlo_module.pb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:59a558236aa7078683af6a702533faa93218604fe964eca709f22952e5376740
3
+ size 9569015
neuronxcc-2.19.8089.0+8ab9f450/MODULE_ef05c748e5821ca55683+253d6470/model.neff ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8027a3f1c7caddb738de49504c75de485814c308672b3bd7eb12ab7674c535fd
3
+ size 17398784