duynht commited on
Commit
db050ed
·
verified ·
1 Parent(s): 8446e96

Upload folder using huggingface_hub

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/checkpoint_metadata.json +18 -0
  2. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/config.yaml +107 -0
  3. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/lr_scheduler/lr_scheduler_pp-0-of-2_tp-0-of-1_exp-0-of-1.pt +3 -0
  4. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/lr_scheduler/lr_scheduler_pp-1-of-2_tp-0-of-1_exp-0-of-1.pt +3 -0
  5. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/0/pp_block/attn/o_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  6. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/0/pp_block/attn/qkv_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  7. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/0/pp_block/input_layernorm/model_weight.safetensors +3 -0
  8. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/0/pp_block/mlp/down_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  9. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/0/pp_block/mlp/gate_up_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  10. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/0/pp_block/post_attention_layernorm/model_weight.safetensors +3 -0
  11. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/1/pp_block/attn/o_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  12. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/1/pp_block/attn/qkv_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  13. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/1/pp_block/input_layernorm/model_weight.safetensors +3 -0
  14. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/1/pp_block/mlp/down_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  15. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/1/pp_block/mlp/gate_up_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  16. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/1/pp_block/post_attention_layernorm/model_weight.safetensors +3 -0
  17. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/10/pp_block/attn/o_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  18. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/10/pp_block/attn/qkv_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  19. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/10/pp_block/input_layernorm/model_weight.safetensors +3 -0
  20. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/10/pp_block/mlp/down_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  21. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/10/pp_block/mlp/gate_up_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  22. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/10/pp_block/post_attention_layernorm/model_weight.safetensors +3 -0
  23. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/11/pp_block/attn/o_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  24. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/11/pp_block/attn/qkv_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  25. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/11/pp_block/input_layernorm/model_weight.safetensors +3 -0
  26. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/11/pp_block/mlp/down_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  27. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/11/pp_block/mlp/gate_up_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  28. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/11/pp_block/post_attention_layernorm/model_weight.safetensors +3 -0
  29. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/12/pp_block/attn/o_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  30. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/12/pp_block/attn/qkv_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  31. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/12/pp_block/input_layernorm/model_weight.safetensors +3 -0
  32. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/12/pp_block/mlp/down_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  33. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/12/pp_block/mlp/gate_up_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  34. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/12/pp_block/post_attention_layernorm/model_weight.safetensors +3 -0
  35. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/13/pp_block/attn/o_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  36. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/13/pp_block/attn/qkv_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  37. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/13/pp_block/input_layernorm/model_weight.safetensors +3 -0
  38. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/13/pp_block/mlp/down_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  39. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/13/pp_block/mlp/gate_up_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  40. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/13/pp_block/post_attention_layernorm/model_weight.safetensors +3 -0
  41. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/14/pp_block/attn/o_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  42. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/14/pp_block/attn/qkv_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  43. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/14/pp_block/input_layernorm/model_weight.safetensors +3 -0
  44. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/14/pp_block/mlp/down_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  45. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/14/pp_block/mlp/gate_up_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  46. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/14/pp_block/post_attention_layernorm/model_weight.safetensors +3 -0
  47. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/15/pp_block/attn/o_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  48. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/15/pp_block/attn/qkv_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
  49. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/15/pp_block/input_layernorm/model_weight.safetensors +3 -0
  50. llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/15/pp_block/mlp/down_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors +3 -0
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/checkpoint_metadata.json ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "custom_metas": null,
3
+ "dp": 2,
4
+ "metas": {
5
+ "consumed_train_samples": 512000,
6
+ "data_stages": [
7
+ {
8
+ "consumed_train_samples": 512000,
9
+ "name": "stable phase",
10
+ "start_training_step": 1
11
+ }
12
+ ],
13
+ "last_stage_idx": 0,
14
+ "last_train_step": 1000
15
+ },
16
+ "tp": 1,
17
+ "version": "1.4"
18
+ }
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/config.yaml ADDED
@@ -0,0 +1,107 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ checkpoints:
2
+ checkpoint_interval: 1000
3
+ checkpoints_path: checkpoints/llama-3.2-3B-finemath-4plus-finetune-bs512-60B-local
4
+ checkpoints_path_is_shared_file_system: false
5
+ load_lr_scheduler: false
6
+ load_optimizer: false
7
+ resume_checkpoint_path: checkpoints/nanotron_pretrained_checkpoints/Nanotron-Llama-3.2-3B
8
+ save_final_state: true
9
+ save_initial_state: false
10
+ data_stages:
11
+ - data:
12
+ dataset:
13
+ dataset_folder:
14
+ - finemath/tokenization/finemath-4-plus-p0.0
15
+ dataset_weights: null
16
+ num_loading_workers: 0
17
+ seed: 8
18
+ name: stable phase
19
+ start_training_step: 1
20
+ general:
21
+ benchmark_csv_path: null
22
+ consumed_train_samples: 512000
23
+ ignore_sanity_checks: true
24
+ project: llama3-3B-CPT
25
+ run: finemath-4plus
26
+ seed: 6
27
+ step: 1000
28
+ lighteval: null
29
+ logging:
30
+ iteration_step_info_interval: 1
31
+ log_level: info
32
+ log_level_replica: info
33
+ model:
34
+ ddp_bucket_cap_mb: 25
35
+ dtype: bfloat16
36
+ init_method:
37
+ std: 0.041666666666666664
38
+ make_vocab_size_divisible_by: 1
39
+ model_config:
40
+ bos_token_id: 128000
41
+ eos_token_id: 128001
42
+ hidden_act: silu
43
+ hidden_size: 3072
44
+ initializer_range: 0.02
45
+ intermediate_size: 8192
46
+ is_llama_config: true
47
+ max_position_embeddings: 4096
48
+ num_attention_heads: 24
49
+ num_hidden_layers: 28
50
+ num_key_value_heads: 8
51
+ pad_token_id: null
52
+ pretraining_tp: 2
53
+ rms_norm_eps: 1.0e-05
54
+ rope_interleaved: false
55
+ rope_scaling:
56
+ factor: 32.0
57
+ high_freq_factor: 4.0
58
+ low_freq_factor: 1.0
59
+ original_max_position_embeddings: 8192
60
+ rope_type: llama3
61
+ rope_theta: 500000.0
62
+ tie_word_embeddings: true
63
+ use_cache: true
64
+ vocab_size: 128256
65
+ optimizer:
66
+ accumulate_grad_in_fp32: true
67
+ clip_grad: 1.0
68
+ learning_rate_scheduler:
69
+ learning_rate: 5.0e-05
70
+ lr_decay_starting_step: 50000
71
+ lr_decay_steps: 10000
72
+ lr_decay_style: linear
73
+ lr_warmup_steps: 1000
74
+ lr_warmup_style: linear
75
+ min_decay_lr: 0
76
+ optimizer_factory:
77
+ adam_beta1: 0.9
78
+ adam_beta2: 0.95
79
+ adam_eps: 1.0e-08
80
+ name: adamW
81
+ torch_adam_is_fused: true
82
+ weight_decay: 0.01
83
+ zero_stage: 1
84
+ parallelism:
85
+ dp: 2
86
+ expert_parallel_size: 1
87
+ pp: 2
88
+ pp_engine: 1f1b
89
+ recompute_layer: true
90
+ tp: 1
91
+ tp_linear_async_communication: true
92
+ tp_mode: REDUCE_SCATTER
93
+ tp_recompute_allgather: true
94
+ profiler: null
95
+ s3_upload: null
96
+ tokenizer:
97
+ tokenizer_max_length: null
98
+ tokenizer_name_or_path: meta-llama/Llama-3.2-3B
99
+ tokenizer_revision: null
100
+ tokens:
101
+ batch_accumulation_per_replica: 32
102
+ limit_test_batches: 0
103
+ limit_val_batches: 0
104
+ micro_batch_size: 8
105
+ sequence_length: 4096
106
+ train_steps: 28612
107
+ val_check_interval: -1
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/lr_scheduler/lr_scheduler_pp-0-of-2_tp-0-of-1_exp-0-of-1.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:64b1904c41c918754e15d16fc0e1250ba24c7d568b08c27ea9438c93bf43e42b
3
+ size 3248
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/lr_scheduler/lr_scheduler_pp-1-of-2_tp-0-of-1_exp-0-of-1.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:01127dadd93637eddd730e5822267d7a3ce61f07f6c03ac9873b3bd091ab23d1
3
+ size 2800
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/0/pp_block/attn/o_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5493674de13ce7e3df04d0d63ff63e66be798c68335816a59ea613ea0ccc1167
3
+ size 18874608
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/0/pp_block/attn/qkv_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:aa86799990558ccb70dbfc61e66689c50920545d9e3f3a6764adb6b32eaa7ecc
3
+ size 31457648
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/0/pp_block/input_layernorm/model_weight.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:59c192c095b876e2b7ca46a6ffd8de0095a18b334f79b5e5e50c5272a1de7186
3
+ size 6240
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/0/pp_block/mlp/down_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0f77148dd47d7f06a7b11def2f0734dd56803a674da56e7d88fb52cde0adb688
3
+ size 50331888
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/0/pp_block/mlp/gate_up_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:de47a53a6b40e494bb634849a5dbd37154419e4c69dcc33adbb949722e956cac
3
+ size 100663608
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/0/pp_block/post_attention_layernorm/model_weight.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:879e858211fac1a5dc714b4ce97e073b987ffdaa61dda9f19873971347ca1bf8
3
+ size 6240
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/1/pp_block/attn/o_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ead6cb8f4d40c5ed60d35fda7a4f5cd16abdf0c172b8f24623b11d7ba4b93817
3
+ size 18874608
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/1/pp_block/attn/qkv_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:52c07408036d849db0dd0ce59a77dc8ef9a904ff5eba85a1b81ce02ded24c87d
3
+ size 31457648
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/1/pp_block/input_layernorm/model_weight.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8254fea25c9490957bd8ea365412df679ce42f4b479fba76a0a40805d03f944f
3
+ size 6240
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/1/pp_block/mlp/down_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:30ed749a776132071616e7619cb1cbf5db19b050e7bb305778fbb5a6181d8c09
3
+ size 50331888
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/1/pp_block/mlp/gate_up_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2b2cf107f041963311af5f8dba437ed0d04b3857cab75e757bfb97bd9e2c6cac
3
+ size 100663608
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/1/pp_block/post_attention_layernorm/model_weight.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fe35fcb4da15a277088739e457da9cf6967a5a3ffcb71bb26b55882ff4fafb7b
3
+ size 6240
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/10/pp_block/attn/o_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ff3bd80a078901359e3fe781f03873e4912568fcfb884a75724e7039d3e6ea76
3
+ size 18874608
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/10/pp_block/attn/qkv_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f3b852d2ca2cd436f56c589befab59f564b4c3d4ce54e45560f02fcfa86753bc
3
+ size 31457648
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/10/pp_block/input_layernorm/model_weight.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e9fb8a20d14a6333756b3d1767f911b18649c3bf6b7f00ff7457e651e0c581f5
3
+ size 6240
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/10/pp_block/mlp/down_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:970e4220d7e388b58243c575ed919de5192c44e7abb5981aaed3bc0ffe8a9c3e
3
+ size 50331888
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/10/pp_block/mlp/gate_up_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:35b6453d359d80a167deed46d2689df359566c730844065584fa3ac4769158d3
3
+ size 100663608
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/10/pp_block/post_attention_layernorm/model_weight.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b76971f8857c2ed4efa0e4a52eaa05306e5e7e0690b280d644e74d80d18da79a
3
+ size 6240
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/11/pp_block/attn/o_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7a349da7992ac30258073cdebdad91a4c4055f78406643837e24b01eb2e0124d
3
+ size 18874608
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/11/pp_block/attn/qkv_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e7e61876d53a05d71b54d61895e3070d2dd158f467762cc07710f52eaad56230
3
+ size 31457648
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/11/pp_block/input_layernorm/model_weight.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eab201b2906337f52643da1625302434ce236b2fbe5a07cc6edb987b6a599903
3
+ size 6240
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/11/pp_block/mlp/down_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:568cdec1f52398f33c96bba87f6445b8c2133d5a55ba137050d5bc6d9f478b28
3
+ size 50331888
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/11/pp_block/mlp/gate_up_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0c0617327ab4f4534272f67d43ef0bbb1f209565bba29187b5b94caccd75141d
3
+ size 100663608
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/11/pp_block/post_attention_layernorm/model_weight.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:58e20f32666687910414d6573c24eec2534fde07f1f233f40b2a10ac7109f749
3
+ size 6240
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/12/pp_block/attn/o_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ca530381500a4a04df3390580ac45fe365b98268114a3f1fb96f999cc0d7fad3
3
+ size 18874608
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/12/pp_block/attn/qkv_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cce4a333e0cf814f6aa0e4aa4c4ad9bce546acc48689ff42771211d417c9f2ce
3
+ size 31457648
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/12/pp_block/input_layernorm/model_weight.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7489902f9b554936da00e2db9b2ee0e0e473c97f122b68cc8aa552773d2505a5
3
+ size 6240
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/12/pp_block/mlp/down_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:22e8f42acb23876dd424d70d4c2c8f84532fdc14b51f672f5d9fb53d8c95ba0b
3
+ size 50331888
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/12/pp_block/mlp/gate_up_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bc41a636d0fb672b49d5b50c596f82291a46957fb25e48492a4d29e2ed1ded1e
3
+ size 100663608
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/12/pp_block/post_attention_layernorm/model_weight.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2f65928096d6874896b8e1a82f5c85d58b0b57f374927cee9aa8228162c57238
3
+ size 6240
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/13/pp_block/attn/o_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5663a2a3ffe435e88be4648ac7dbafd1b20a56068fcd1402674a7998f61215e9
3
+ size 18874608
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/13/pp_block/attn/qkv_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:82c6cbc3afe11e568e4ec14c05d7b30d6d8cd23ef3027713e9f3d196d7c856f5
3
+ size 31457648
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/13/pp_block/input_layernorm/model_weight.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a0a89b383e9b081be8456e4e19ee71f8c830fa41c0a9d7959555cbd37408f1e9
3
+ size 6240
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/13/pp_block/mlp/down_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4b19cec7cae68d67a0b21eb8b4d676cc567e3b633300b91c76af7d925f92d3bd
3
+ size 50331888
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/13/pp_block/mlp/gate_up_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d226d99e5f1a16c5e0ac6b751342de343678070b796f8aad6b98df5aba6e174e
3
+ size 100663608
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/13/pp_block/post_attention_layernorm/model_weight.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4d5700ed85925816302d8c8c1f8a0897ed40fab88a0543043a4bc9d0a0feb892
3
+ size 6240
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/14/pp_block/attn/o_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a80aa90e167e1cab4b9c580e2acc28ce655857cb3ca8a76b7c4120ebaa607103
3
+ size 18874608
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/14/pp_block/attn/qkv_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c21335f74c5f5ecc31dfee4e9a9e2c4683428289d738db5dfa45c975e565923a
3
+ size 31457648
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/14/pp_block/input_layernorm/model_weight.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:222134760b958d6ca7022fc6f7a9f299311afa2e99b65502c379fc73db2e6596
3
+ size 6240
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/14/pp_block/mlp/down_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d6a17fcc580b691a7efc41bc9bda98b5689d6b4d00e1df29f2767165a2796be9
3
+ size 50331888
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/14/pp_block/mlp/gate_up_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cd5db9eaaf927f14d1ef2f91b94a709194543e89d140812b2fed94faca896257
3
+ size 100663608
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/14/pp_block/post_attention_layernorm/model_weight.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5decd84a8eff9d0394597070333b6faf03878dda3eab2145f81bb7426a6739f8
3
+ size 6240
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/15/pp_block/attn/o_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:54f03bee6326767a1d39dea40155e6f4c1dbfd84d25a603d54ee69fcaf91e40c
3
+ size 18874608
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/15/pp_block/attn/qkv_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8d9cdec71c9db7e953f1fd7705024115346d92a7313028e78c3e62630503d28f
3
+ size 31457648
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/15/pp_block/input_layernorm/model_weight.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2d18fa9dec303593a03dd8404e793fbd5b5266e4b7941990e7d539d7a909968d
3
+ size 6240
llama-3.2-3B-finemath-4plus-p0.0-bs512-tc60B/1000/model/model/decoder/15/pp_block/mlp/down_proj/model_weight_pp-rank-0-of-2_tp-rank-0-of-1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:456d75fb3c0bc293ccd8d6a508bc9df708ce5edc437e18cb5d9f207497ca0f85
3
+ size 50331888