Upload folder using huggingface_hub

Browse files

Files changed (16) hide show

arc_challenge_25shot_bs16_bf16.json +25 -0
config.json +28 -0
generation_config.json +6 -0
gsm8k_5shot_bs16_bf16.json +23 -0
hellaswag_10shot_bs16_bf16.json +25 -0
mmlu_5shot_bs4_bf16.json +417 -0
pytorch_model-00001-of-00003.bin +3 -0
pytorch_model-00002-of-00003.bin +3 -0
pytorch_model-00003-of-00003.bin +3 -0
pytorch_model.bin.index.json +298 -0
special_tokens_map.json +23 -0
tokenizer.json +0 -0
tokenizer.model +3 -0
tokenizer_config.json +41 -0
truthfulqa_mc_0shot_bs16_bf16.json +25 -0
winogrande_5shot_bs16_bf16.json +23 -0

arc_challenge_25shot_bs16_bf16.json ADDED Viewed

	@@ -0,0 +1,25 @@

+{
+  "results": {
+    "arc_challenge": {
+      "acc": 0.4667235494880546,
+      "acc_stderr": 0.01457899585960581,
+      "acc_norm": 0.48976109215017066,
+      "acc_norm_stderr": 0.014608326906285019
+    }
+  },
+  "versions": {
+    "arc_challenge": 0
+  },
+  "config": {
+    "model": "sparseml",
+    "model_args": "pretrained=/network/alexandre/research/cerebras/llama2_7B_sparse50_45B_retrained/ultrachat200k/llama2_7B_45B_sparse50_LR2e-4_GC2_E2/training,dtype=bfloat16",
+    "num_fewshot": 25,
+    "batch_size": "16",
+    "batch_sizes": [],
+    "device": "cuda:1",
+    "no_cache": true,
+    "limit": null,
+    "bootstrap_iters": 100000,
+    "description_dict": {}
+  }
+}

config.json ADDED Viewed

	@@ -0,0 +1,28 @@

+{
+  "architectures": [
+    "LlamaForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "bos_token_id": 1,
+  "eos_token_id": 2,
+  "hidden_act": "silu",
+  "hidden_size": 4096,
+  "initializer_range": 0.02,
+  "intermediate_size": 11008,
+  "max_position_embeddings": 4096,
+  "model_type": "llama",
+  "num_attention_heads": 32,
+  "num_hidden_layers": 32,
+  "num_key_value_heads": 32,
+  "pretraining_tp": 1,
+  "rms_norm_eps": 1e-05,
+  "rope_scaling": null,
+  "rope_theta": 10000.0,
+  "tie_word_embeddings": false,
+  "tokenizer_class": "LlamaTokenizerFast",
+  "torch_dtype": "float32",
+  "transformers_version": "1.7.0.20240308",
+  "use_cache": true,
+  "vocab_size": 32000
+}

generation_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 1,
+  "eos_token_id": 2,
+  "transformers_version": "1.7.0.20240308"
+}

gsm8k_5shot_bs16_bf16.json ADDED Viewed

	@@ -0,0 +1,23 @@

+{
+  "results": {
+    "gsm8k": {
+      "acc": 0.07960576194086429,
+      "acc_stderr": 0.007455924338676284
+    }
+  },
+  "versions": {
+    "gsm8k": 0
+  },
+  "config": {
+    "model": "sparseml",
+    "model_args": "pretrained=/network/alexandre/research/cerebras/llama2_7B_sparse50_45B_retrained/ultrachat200k/llama2_7B_45B_sparse50_LR2e-4_GC2_E2/training,dtype=bfloat16",
+    "num_fewshot": 5,
+    "batch_size": "16",
+    "batch_sizes": [],
+    "device": "cuda:6",
+    "no_cache": true,
+    "limit": null,
+    "bootstrap_iters": 100000,
+    "description_dict": {}
+  }
+}

hellaswag_10shot_bs16_bf16.json ADDED Viewed

	@@ -0,0 +1,25 @@

+{
+  "results": {
+    "hellaswag": {
+      "acc": 0.5541724756024696,
+      "acc_stderr": 0.004960408362133245,
+      "acc_norm": 0.7353116908982275,
+      "acc_norm_stderr": 0.00440265476726963
+    }
+  },
+  "versions": {
+    "hellaswag": 0
+  },
+  "config": {
+    "model": "sparseml",
+    "model_args": "pretrained=/network/alexandre/research/cerebras/llama2_7B_sparse50_45B_retrained/ultrachat200k/llama2_7B_45B_sparse50_LR2e-4_GC2_E2/training,dtype=bfloat16",
+    "num_fewshot": 10,
+    "batch_size": "16",
+    "batch_sizes": [],
+    "device": "cuda:6",
+    "no_cache": true,
+    "limit": null,
+    "bootstrap_iters": 100000,
+    "description_dict": {}
+  }
+}

mmlu_5shot_bs4_bf16.json ADDED Viewed

	@@ -0,0 +1,417 @@

+{
+  "results": {
+    "hendrycksTest-abstract_algebra": {
+      "acc": 0.35,
+      "acc_stderr": 0.0479372485441102,
+      "acc_norm": 0.35,
+      "acc_norm_stderr": 0.0479372485441102
+    },
+    "hendrycksTest-anatomy": {
+      "acc": 0.4,
+      "acc_stderr": 0.04232073695151589,
+      "acc_norm": 0.4,
+      "acc_norm_stderr": 0.04232073695151589
+    },
+    "hendrycksTest-astronomy": {
+      "acc": 0.3684210526315789,
+      "acc_stderr": 0.03925523381052932,
+      "acc_norm": 0.3684210526315789,
+      "acc_norm_stderr": 0.03925523381052932
+    },
+    "hendrycksTest-business_ethics": {
+      "acc": 0.43,
+      "acc_stderr": 0.049756985195624284,
+      "acc_norm": 0.43,
+      "acc_norm_stderr": 0.049756985195624284
+    },
+    "hendrycksTest-clinical_knowledge": {
+      "acc": 0.42641509433962266,
+      "acc_stderr": 0.030437794342983045,
+      "acc_norm": 0.42641509433962266,
+      "acc_norm_stderr": 0.030437794342983045
+    },
+    "hendrycksTest-college_biology": {
+      "acc": 0.3680555555555556,
+      "acc_stderr": 0.040329990539607195,
+      "acc_norm": 0.3680555555555556,
+      "acc_norm_stderr": 0.040329990539607195
+    },
+    "hendrycksTest-college_chemistry": {
+      "acc": 0.3,
+      "acc_stderr": 0.046056618647183814,
+      "acc_norm": 0.3,
+      "acc_norm_stderr": 0.046056618647183814
+    },
+    "hendrycksTest-college_computer_science": {
+      "acc": 0.38,
+      "acc_stderr": 0.048783173121456316,
+      "acc_norm": 0.38,
+      "acc_norm_stderr": 0.048783173121456316
+    },
+    "hendrycksTest-college_mathematics": {
+      "acc": 0.31,
+      "acc_stderr": 0.04648231987117316,
+      "acc_norm": 0.31,
+      "acc_norm_stderr": 0.04648231987117316
+    },
+    "hendrycksTest-college_medicine": {
+      "acc": 0.3063583815028902,
+      "acc_stderr": 0.03514942551267437,
+      "acc_norm": 0.3063583815028902,
+      "acc_norm_stderr": 0.03514942551267437
+    },
+    "hendrycksTest-college_physics": {
+      "acc": 0.23529411764705882,
+      "acc_stderr": 0.04220773659171453,
+      "acc_norm": 0.23529411764705882,
+      "acc_norm_stderr": 0.04220773659171453
+    },
+    "hendrycksTest-computer_security": {
+      "acc": 0.55,
+      "acc_stderr": 0.05,
+      "acc_norm": 0.55,
+      "acc_norm_stderr": 0.05
+    },
+    "hendrycksTest-conceptual_physics": {
+      "acc": 0.3617021276595745,
+      "acc_stderr": 0.03141082197596239,
+      "acc_norm": 0.3617021276595745,
+      "acc_norm_stderr": 0.03141082197596239
+    },
+    "hendrycksTest-econometrics": {
+      "acc": 0.2982456140350877,
+      "acc_stderr": 0.043036840335373146,
+      "acc_norm": 0.2982456140350877,
+      "acc_norm_stderr": 0.043036840335373146
+    },
+    "hendrycksTest-electrical_engineering": {
+      "acc": 0.4,
+      "acc_stderr": 0.04082482904638629,
+      "acc_norm": 0.4,
+      "acc_norm_stderr": 0.04082482904638629
+    },
+    "hendrycksTest-elementary_mathematics": {
+      "acc": 0.2698412698412698,
+      "acc_stderr": 0.02286083830923207,
+      "acc_norm": 0.2698412698412698,
+      "acc_norm_stderr": 0.02286083830923207
+    },
+    "hendrycksTest-formal_logic": {
+      "acc": 0.2777777777777778,
+      "acc_stderr": 0.04006168083848876,
+      "acc_norm": 0.2777777777777778,
+      "acc_norm_stderr": 0.04006168083848876
+    },
+    "hendrycksTest-global_facts": {
+      "acc": 0.29,
+      "acc_stderr": 0.04560480215720683,
+      "acc_norm": 0.29,
+      "acc_norm_stderr": 0.04560480215720683
+    },
+    "hendrycksTest-high_school_biology": {
+      "acc": 0.43870967741935485,
+      "acc_stderr": 0.028229497320317213,
+      "acc_norm": 0.43870967741935485,
+      "acc_norm_stderr": 0.028229497320317213
+    },
+    "hendrycksTest-high_school_chemistry": {
+      "acc": 0.27586206896551724,
+      "acc_stderr": 0.03144712581678241,
+      "acc_norm": 0.27586206896551724,
+      "acc_norm_stderr": 0.03144712581678241
+    },
+    "hendrycksTest-high_school_computer_science": {
+      "acc": 0.42,
+      "acc_stderr": 0.049604496374885836,
+      "acc_norm": 0.42,
+      "acc_norm_stderr": 0.049604496374885836
+    },
+    "hendrycksTest-high_school_european_history": {
+      "acc": 0.5636363636363636,
+      "acc_stderr": 0.03872592983524754,
+      "acc_norm": 0.5636363636363636,
+      "acc_norm_stderr": 0.03872592983524754
+    },
+    "hendrycksTest-high_school_geography": {
+      "acc": 0.4494949494949495,
+      "acc_stderr": 0.0354413249194797,
+      "acc_norm": 0.4494949494949495,
+      "acc_norm_stderr": 0.0354413249194797
+    },
+    "hendrycksTest-high_school_government_and_politics": {
+      "acc": 0.533678756476684,
+      "acc_stderr": 0.036002440698671784,
+      "acc_norm": 0.533678756476684,
+      "acc_norm_stderr": 0.036002440698671784
+    },
+    "hendrycksTest-high_school_macroeconomics": {
+      "acc": 0.3230769230769231,
+      "acc_stderr": 0.023710888501970562,
+      "acc_norm": 0.3230769230769231,
+      "acc_norm_stderr": 0.023710888501970562
+    },
+    "hendrycksTest-high_school_mathematics": {
+      "acc": 0.24814814814814815,
+      "acc_stderr": 0.0263357394040558,
+      "acc_norm": 0.24814814814814815,
+      "acc_norm_stderr": 0.0263357394040558
+    },
+    "hendrycksTest-high_school_microeconomics": {
+      "acc": 0.3739495798319328,
+      "acc_stderr": 0.03142946637883708,
+      "acc_norm": 0.3739495798319328,
+      "acc_norm_stderr": 0.03142946637883708
+    },
+    "hendrycksTest-high_school_physics": {
+      "acc": 0.304635761589404,
+      "acc_stderr": 0.037579499229433426,
+      "acc_norm": 0.304635761589404,
+      "acc_norm_stderr": 0.037579499229433426
+    },
+    "hendrycksTest-high_school_psychology": {
+      "acc": 0.4990825688073395,
+      "acc_stderr": 0.021437287056051215,
+      "acc_norm": 0.4990825688073395,
+      "acc_norm_stderr": 0.021437287056051215
+    },
+    "hendrycksTest-high_school_statistics": {
+      "acc": 0.27314814814814814,
+      "acc_stderr": 0.030388051301678116,
+      "acc_norm": 0.27314814814814814,
+      "acc_norm_stderr": 0.030388051301678116
+    },
+    "hendrycksTest-high_school_us_history": {
+      "acc": 0.5147058823529411,
+      "acc_stderr": 0.03507793834791323,
+      "acc_norm": 0.5147058823529411,
+      "acc_norm_stderr": 0.03507793834791323
+    },
+    "hendrycksTest-high_school_world_history": {
+      "acc": 0.5611814345991561,
+      "acc_stderr": 0.032302649315470375,
+      "acc_norm": 0.5611814345991561,
+      "acc_norm_stderr": 0.032302649315470375
+    },
+    "hendrycksTest-human_aging": {
+      "acc": 0.47085201793721976,
+      "acc_stderr": 0.03350073248773404,
+      "acc_norm": 0.47085201793721976,
+      "acc_norm_stderr": 0.03350073248773404
+    },
+    "hendrycksTest-human_sexuality": {
+      "acc": 0.45038167938931295,
+      "acc_stderr": 0.04363643698524779,
+      "acc_norm": 0.45038167938931295,
+      "acc_norm_stderr": 0.04363643698524779
+    },
+    "hendrycksTest-international_law": {
+      "acc": 0.628099173553719,
+      "acc_stderr": 0.04412015806624504,
+      "acc_norm": 0.628099173553719,
+      "acc_norm_stderr": 0.04412015806624504
+    },
+    "hendrycksTest-jurisprudence": {
+      "acc": 0.4074074074074074,
+      "acc_stderr": 0.04750077341199986,
+      "acc_norm": 0.4074074074074074,
+      "acc_norm_stderr": 0.04750077341199986
+    },
+    "hendrycksTest-logical_fallacies": {
+      "acc": 0.48466257668711654,
+      "acc_stderr": 0.03926522378708843,
+      "acc_norm": 0.48466257668711654,
+      "acc_norm_stderr": 0.03926522378708843
+    },
+    "hendrycksTest-machine_learning": {
+      "acc": 0.4107142857142857,
+      "acc_stderr": 0.04669510663875191,
+      "acc_norm": 0.4107142857142857,
+      "acc_norm_stderr": 0.04669510663875191
+    },
+    "hendrycksTest-management": {
+      "acc": 0.4563106796116505,
+      "acc_stderr": 0.049318019942204146,
+      "acc_norm": 0.4563106796116505,
+      "acc_norm_stderr": 0.049318019942204146
+    },
+    "hendrycksTest-marketing": {
+      "acc": 0.6367521367521367,
+      "acc_stderr": 0.03150712523091264,
+      "acc_norm": 0.6367521367521367,
+      "acc_norm_stderr": 0.03150712523091264
+    },
+    "hendrycksTest-medical_genetics": {
+      "acc": 0.43,
+      "acc_stderr": 0.049756985195624284,
+      "acc_norm": 0.43,
+      "acc_norm_stderr": 0.049756985195624284
+    },
+    "hendrycksTest-miscellaneous": {
+      "acc": 0.5606641123882503,
+      "acc_stderr": 0.017747874245683606,
+      "acc_norm": 0.5606641123882503,
+      "acc_norm_stderr": 0.017747874245683606
+    },
+    "hendrycksTest-moral_disputes": {
+      "acc": 0.47109826589595377,
+      "acc_stderr": 0.026874085883518348,
+      "acc_norm": 0.47109826589595377,
+      "acc_norm_stderr": 0.026874085883518348
+    },
+    "hendrycksTest-moral_scenarios": {
+      "acc": 0.23798882681564246,
+      "acc_stderr": 0.014242630070574915,
+      "acc_norm": 0.23798882681564246,
+      "acc_norm_stderr": 0.014242630070574915
+    },
+    "hendrycksTest-nutrition": {
+      "acc": 0.434640522875817,
+      "acc_stderr": 0.028384256704883037,
+      "acc_norm": 0.434640522875817,
+      "acc_norm_stderr": 0.028384256704883037
+    },
+    "hendrycksTest-philosophy": {
+      "acc": 0.4758842443729904,
+      "acc_stderr": 0.028365041542564577,
+      "acc_norm": 0.4758842443729904,
+      "acc_norm_stderr": 0.028365041542564577
+    },
+    "hendrycksTest-prehistory": {
+      "acc": 0.4537037037037037,
+      "acc_stderr": 0.027701228468542602,
+      "acc_norm": 0.4537037037037037,
+      "acc_norm_stderr": 0.027701228468542602
+    },
+    "hendrycksTest-professional_accounting": {
+      "acc": 0.32978723404255317,
+      "acc_stderr": 0.028045946942042398,
+      "acc_norm": 0.32978723404255317,
+      "acc_norm_stderr": 0.028045946942042398
+    },
+    "hendrycksTest-professional_law": {
+      "acc": 0.3494132985658409,
+      "acc_stderr": 0.012177306252786686,
+      "acc_norm": 0.3494132985658409,
+      "acc_norm_stderr": 0.012177306252786686
+    },
+    "hendrycksTest-professional_medicine": {
+      "acc": 0.39705882352941174,
+      "acc_stderr": 0.029722152099280058,
+      "acc_norm": 0.39705882352941174,
+      "acc_norm_stderr": 0.029722152099280058
+    },
+    "hendrycksTest-professional_psychology": {
+      "acc": 0.41013071895424835,
+      "acc_stderr": 0.019898412717635913,
+      "acc_norm": 0.41013071895424835,
+      "acc_norm_stderr": 0.019898412717635913
+    },
+    "hendrycksTest-public_relations": {
+      "acc": 0.4727272727272727,
+      "acc_stderr": 0.04782001791380063,
+      "acc_norm": 0.4727272727272727,
+      "acc_norm_stderr": 0.04782001791380063
+    },
+    "hendrycksTest-security_studies": {
+      "acc": 0.4163265306122449,
+      "acc_stderr": 0.03155782816556164,
+      "acc_norm": 0.4163265306122449,
+      "acc_norm_stderr": 0.03155782816556164
+    },
+    "hendrycksTest-sociology": {
+      "acc": 0.5422885572139303,
+      "acc_stderr": 0.03522865864099597,
+      "acc_norm": 0.5422885572139303,
+      "acc_norm_stderr": 0.03522865864099597
+    },
+    "hendrycksTest-us_foreign_policy": {
+      "acc": 0.56,
+      "acc_stderr": 0.04988876515698589,
+      "acc_norm": 0.56,
+      "acc_norm_stderr": 0.04988876515698589
+    },
+    "hendrycksTest-virology": {
+      "acc": 0.4397590361445783,
+      "acc_stderr": 0.03864139923699121,
+      "acc_norm": 0.4397590361445783,
+      "acc_norm_stderr": 0.03864139923699121
+    },
+    "hendrycksTest-world_religions": {
+      "acc": 0.5789473684210527,
+      "acc_stderr": 0.03786720706234214,
+      "acc_norm": 0.5789473684210527,
+      "acc_norm_stderr": 0.03786720706234214
+    }
+  },
+  "versions": {
+    "hendrycksTest-abstract_algebra": 1,
+    "hendrycksTest-anatomy": 1,
+    "hendrycksTest-astronomy": 1,
+    "hendrycksTest-business_ethics": 1,
+    "hendrycksTest-clinical_knowledge": 1,
+    "hendrycksTest-college_biology": 1,
+    "hendrycksTest-college_chemistry": 1,
+    "hendrycksTest-college_computer_science": 1,
+    "hendrycksTest-college_mathematics": 1,
+    "hendrycksTest-college_medicine": 1,
+    "hendrycksTest-college_physics": 1,
+    "hendrycksTest-computer_security": 1,
+    "hendrycksTest-conceptual_physics": 1,
+    "hendrycksTest-econometrics": 1,
+    "hendrycksTest-electrical_engineering": 1,
+    "hendrycksTest-elementary_mathematics": 1,
+    "hendrycksTest-formal_logic": 1,
+    "hendrycksTest-global_facts": 1,
+    "hendrycksTest-high_school_biology": 1,
+    "hendrycksTest-high_school_chemistry": 1,
+    "hendrycksTest-high_school_computer_science": 1,
+    "hendrycksTest-high_school_european_history": 1,
+    "hendrycksTest-high_school_geography": 1,
+    "hendrycksTest-high_school_government_and_politics": 1,
+    "hendrycksTest-high_school_macroeconomics": 1,
+    "hendrycksTest-high_school_mathematics": 1,
+    "hendrycksTest-high_school_microeconomics": 1,
+    "hendrycksTest-high_school_physics": 1,
+    "hendrycksTest-high_school_psychology": 1,
+    "hendrycksTest-high_school_statistics": 1,
+    "hendrycksTest-high_school_us_history": 1,
+    "hendrycksTest-high_school_world_history": 1,
+    "hendrycksTest-human_aging": 1,
+    "hendrycksTest-human_sexuality": 1,
+    "hendrycksTest-international_law": 1,
+    "hendrycksTest-jurisprudence": 1,
+    "hendrycksTest-logical_fallacies": 1,
+    "hendrycksTest-machine_learning": 1,
+    "hendrycksTest-management": 1,
+    "hendrycksTest-marketing": 1,
+    "hendrycksTest-medical_genetics": 1,
+    "hendrycksTest-miscellaneous": 1,
+    "hendrycksTest-moral_disputes": 1,
+    "hendrycksTest-moral_scenarios": 1,
+    "hendrycksTest-nutrition": 1,
+    "hendrycksTest-philosophy": 1,
+    "hendrycksTest-prehistory": 1,
+    "hendrycksTest-professional_accounting": 1,
+    "hendrycksTest-professional_law": 1,
+    "hendrycksTest-professional_medicine": 1,
+    "hendrycksTest-professional_psychology": 1,
+    "hendrycksTest-public_relations": 1,
+    "hendrycksTest-security_studies": 1,
+    "hendrycksTest-sociology": 1,
+    "hendrycksTest-us_foreign_policy": 1,
+    "hendrycksTest-virology": 1,
+    "hendrycksTest-world_religions": 1
+  },
+  "config": {
+    "model": "sparseml",
+    "model_args": "pretrained=/network/alexandre/research/cerebras/llama2_7B_sparse50_45B_retrained/ultrachat200k/llama2_7B_45B_sparse50_LR2e-4_GC2_E2/training,dtype=bfloat16",
+    "num_fewshot": 5,
+    "batch_size": "4",
+    "batch_sizes": [],
+    "device": "cuda:6",
+    "no_cache": true,
+    "limit": null,
+    "bootstrap_iters": 100000,
+    "description_dict": {}
+  }
+}

pytorch_model-00001-of-00003.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:fcc85ea17c5298ee5b950590585c1160486da2499c7a01646e84d9a076404133
+size 9877982873

pytorch_model-00002-of-00003.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e57c4db65062e4ab74489af6bd9a20d6ec4449d1d4732c71debc3246761f884a
+size 9894794253

pytorch_model-00003-of-00003.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e467d07348e67aeef4c85265cda863a7568f5782e564adbc0361ef52df86a3bf
+size 7180986412

pytorch_model.bin.index.json ADDED Viewed

	@@ -0,0 +1,298 @@

+{
+  "metadata": {
+    "total_size": 26953662464
+  },
+  "weight_map": {
+    "lm_head.weight": "pytorch_model-00003-of-00003.bin",
+    "model.embed_tokens.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.0.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.0.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.0.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.0.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.0.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.0.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.0.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.0.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.0.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.1.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.1.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.1.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.1.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.1.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.1.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.1.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.1.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.1.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.10.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.10.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.10.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.10.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.10.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.10.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.10.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.10.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.10.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.11.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.11.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.11.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.11.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.11.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.11.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.11.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.11.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.11.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.12.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.12.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.12.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.12.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.12.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.12.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.12.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.12.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.12.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.13.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.13.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.13.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.13.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.13.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.13.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.13.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.13.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.13.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.14.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.14.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.14.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.14.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.14.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.14.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.14.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.14.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.14.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.15.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.15.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.15.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.15.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.15.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.15.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.15.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.15.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.15.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.16.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.16.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.16.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.16.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.16.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.16.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.16.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.16.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.16.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.17.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.17.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.17.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.17.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.17.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.17.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.17.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.17.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.17.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.18.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.18.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.18.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.18.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.18.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.18.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.18.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.18.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.18.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.19.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.19.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.19.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.19.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.19.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.19.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.19.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.19.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.19.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.2.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.2.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.2.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.2.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.2.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.2.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.2.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.2.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.2.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.20.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.20.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.20.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.20.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.20.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.20.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.20.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.20.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.20.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.21.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.21.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.21.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.21.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.21.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.21.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.21.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.21.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.21.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.22.input_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.22.mlp.down_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.22.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.22.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.22.post_attention_layernorm.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.22.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.22.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.22.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.22.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.23.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.23.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.23.mlp.gate_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.23.mlp.up_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.23.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.23.self_attn.k_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.23.self_attn.o_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.23.self_attn.q_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.23.self_attn.v_proj.weight": "pytorch_model-00002-of-00003.bin",
+    "model.layers.24.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.24.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.24.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.24.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.24.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.24.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.24.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.24.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.24.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.25.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.25.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.25.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.25.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.25.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.25.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.25.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.25.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.25.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.26.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.26.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.26.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.26.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.26.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.26.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.26.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.26.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.26.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.27.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.27.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.27.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.27.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.27.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.27.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.27.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.27.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.27.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.28.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.28.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.28.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.28.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.28.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.28.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.28.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.28.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.28.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.29.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.29.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.29.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.29.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.29.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.29.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.29.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.29.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.29.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.3.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.3.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.3.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.3.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.3.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.3.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.3.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.3.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.3.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.30.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.30.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.30.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.30.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.30.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.30.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.30.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.30.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.30.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.31.input_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.31.mlp.down_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.31.mlp.gate_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.31.mlp.up_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.31.post_attention_layernorm.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.31.self_attn.k_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.31.self_attn.o_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.31.self_attn.q_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.31.self_attn.v_proj.weight": "pytorch_model-00003-of-00003.bin",
+    "model.layers.4.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.4.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.4.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.4.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.4.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.4.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.4.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.4.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.4.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.5.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.5.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.5.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.5.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.5.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.5.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.5.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.5.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.5.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.6.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.6.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.6.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.6.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.6.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.6.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.6.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.6.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.6.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.7.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.7.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.7.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.7.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.7.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.7.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.7.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.7.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.7.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.8.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.8.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.8.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.8.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.8.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.8.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.8.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.8.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.8.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.9.input_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.9.mlp.down_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.9.mlp.gate_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.9.mlp.up_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.9.post_attention_layernorm.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.9.self_attn.k_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.9.self_attn.o_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.9.self_attn.q_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.layers.9.self_attn.v_proj.weight": "pytorch_model-00001-of-00003.bin",
+    "model.norm.weight": "pytorch_model-00003-of-00003.bin"
+  }
+}

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,23 @@

+{
+  "bos_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer.model ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9e556afd44213b6bd1be2b850ebbbd98f5481437a8021afaf58ee7fb1818d347
+size 499723

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,41 @@

+{
+  "add_bos_token": true,
+  "add_eos_token": false,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<s>",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "</s>",
+  "legacy": false,
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": null,
+  "padding_side": "right",
+  "sp_model_kwargs": {},
+  "tokenizer_class": "LlamaTokenizer",
+  "unk_token": "<unk>",
+  "use_default_system_prompt": false
+}

truthfulqa_mc_0shot_bs16_bf16.json ADDED Viewed

	@@ -0,0 +1,25 @@

+{
+  "results": {
+    "truthfulqa_mc": {
+      "mc1": 0.2631578947368421,
+      "mc1_stderr": 0.015415241740237012,
+      "mc2": 0.39511662473536746,
+      "mc2_stderr": 0.014932053345393062
+    }
+  },
+  "versions": {
+    "truthfulqa_mc": 1
+  },
+  "config": {
+    "model": "sparseml",
+    "model_args": "pretrained=/network/alexandre/research/cerebras/llama2_7B_sparse50_45B_retrained/ultrachat200k/llama2_7B_45B_sparse50_LR2e-4_GC2_E2/training,dtype=bfloat16",
+    "num_fewshot": 0,
+    "batch_size": "16",
+    "batch_sizes": [],
+    "device": "cuda:4",
+    "no_cache": true,
+    "limit": null,
+    "bootstrap_iters": 100000,
+    "description_dict": {}
+  }
+}

winogrande_5shot_bs16_bf16.json ADDED Viewed

	@@ -0,0 +1,23 @@

+{
+  "results": {
+    "winogrande": {
+      "acc": 0.6779794790844514,
+      "acc_stderr": 0.01313207020207106
+    }
+  },
+  "versions": {
+    "winogrande": 0
+  },
+  "config": {
+    "model": "sparseml",
+    "model_args": "pretrained=/network/alexandre/research/cerebras/llama2_7B_sparse50_45B_retrained/ultrachat200k/llama2_7B_45B_sparse50_LR2e-4_GC2_E2/training,dtype=bfloat16",
+    "num_fewshot": 5,
+    "batch_size": "16",
+    "batch_sizes": [],
+    "device": "cuda:2",
+    "no_cache": true,
+    "limit": null,
+    "bootstrap_iters": 100000,
+    "description_dict": {}
+  }
+}